From 96afc351aced47bd9dd995c6b4a77d79b3284f32 Mon Sep 17 00:00:00 2001 From: TheMajesticMagician Date: Thu, 30 Jul 2026 20:50:41 -0600 Subject: [PATCH] Add text-to-dialogue, self-restart capability, and misc updates --- .claude/settings.local.json | 33 ++- .env.example | 30 +++ CLAUDE.md | 99 ++++++++- README.md | 43 +++- bolt_pet/audio/tts.py | 108 ++++++++-- bolt_pet/config.py | 39 +++- bolt_pet/controller.py | 217 +++++++++++++++++++- bolt_pet/dialogue.py | 219 ++++++++++++++++++++ bolt_pet/pet_actions.py | 26 ++- bolt_pet/self_restart.py | 235 ++++++++++++++++++++++ bolt_pet/server_client.py | 26 ++- bolt_pet/state.py | 7 +- bolt_pet/ui/app.py | 4 + bolt_pet/ui/tray.py | 24 ++- tests/test_controller.py | 6 +- tests/test_controller_features.py | 320 ++++++++++++++++++++++++++++-- tests/test_dialogue.py | 225 +++++++++++++++++++++ tests/test_pet_actions.py | 14 ++ tests/test_self_restart.py | 158 +++++++++++++++ tests/test_server_client.py | 18 +- tests/test_tts_stream.py | 27 ++- 21 files changed, 1819 insertions(+), 59 deletions(-) create mode 100644 bolt_pet/dialogue.py create mode 100644 bolt_pet/self_restart.py create mode 100644 tests/test_dialogue.py create mode 100644 tests/test_self_restart.py diff --git a/.claude/settings.local.json b/.claude/settings.local.json index 13bbabb..1bf98b5 100644 --- a/.claude/settings.local.json +++ b/.claude/settings.local.json @@ -52,7 +52,38 @@ "Bash(docker exec bolt *)", "Bash(QT_QPA_PLATFORM=offscreen /root/Documents/bolt-pet/.venv/bin/pytest /home/themajesticmagician/Documents/Bolt-Pet/tests/test_file_ops.py -q)", "Bash(grep -rn *)", - "Bash(git ls-tree *)" + "Bash(git ls-tree *)", + "Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_controller_features.py -q)", + "Read(//home/maji/Documents/tmn-api/**)", + "Bash(timeout 300 .venv/bin/python -m pytest tests/test_desk_voice.py -q)", + "Bash(echo \"exit=$?\")", + "Bash(timeout 300 /home/maji/Documents/tmn-api/.venv/bin/python -m pytest /home/maji/Documents/tmn-api/tests/test_desk_voice.py -q -p no:cacheprovider --rootdir=/home/maji/Documents/tmn-api)", + "Bash(echo \"EXIT=$?\")", + "Bash(/home/maji/Documents/tmn-api/.venv/bin/python -c \"import ast,pathlib; ast.parse\\(pathlib.Path\\('/home/maji/Documents/tmn-api/ai/desk_api.py'\\).read_text\\(\\)\\); print\\('desk_api.py parses OK'\\)\")", + "Bash(ps -eo pid,etime,cmd)", + "Bash(systemctl --user list-units --type=service)", + "Read(//run/user/1000/gvfs/sftp:host=192.168.2.231,user=root/Main/Docker-Compose/TMN-API/tmn-api/ai/**)", + "Bash(findmnt -T /home/maji/Documents/tmn-api -o TARGET,SOURCE,FSTYPE)", + "Bash(echo \"rc=$?\")", + "Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py -q -p no:cacheprovider)", + "Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py -q -p no:cacheprovider)", + "Bash(awk 'NR>=1150 && NR<=1310 && \\(/return / || /def /\\)' ai/agents/default.py)", + "Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python -c ' *)", + "Bash(timeout 900 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py -q -p no:cacheprovider)", + "Bash(timeout 300 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_proactive.py::test_send_trims_swallowed_tool_lines -q -p no:cacheprovider)", + "Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python *)", + "Bash(./tmnvenv/bin/pip install *)", + "Bash(./tmnvenv/bin/python -c \"import pytest,dotenv,yaml; print\\('scratch venv ready'\\)\")", + "Bash(/tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/pip install *)", + "Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py)", + "Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider)", + "Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py)", + "Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_default_agent.py -q -p no:cacheprovider)", + "Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py -q -p no:cacheprovider)", + "Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py tests/test_desk_api.py tests/test_main_helpers.py -q -p no:cacheprovider)", + "Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py --ignore=tests/test_billing_web.py)", + "WebFetch(domain:elevenlabs.io)", + "Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_dialogue.py -q)" ] } } diff --git a/.env.example b/.env.example index f2a632f..c0da76b 100644 --- a/.env.example +++ b/.env.example @@ -31,6 +31,36 @@ ELEVENLABS_VOICE_ID= #ELEVENLABS_MODEL_ID=eleven_flash_v2 #TTS_SAMPLE_RATE=24000 +# Ask Bolt to use a different voice (or another language) and the server +# picks one from the ElevenLabs voice library and tags the reply with it. +# It only tags one reply, and it can't remember the id afterwards — so the +# pet keeps using that voice until a new one is picked or you choose "Use +# default voice" in the tray. VOICE_STICKY=false makes each pick last for +# exactly the one reply it came with instead. +#VOICE_STICKY=true + +# ── Multi-voice dialogue (ElevenLabs Text to Dialogue) ────────────────────── +# Lets Bolt play a short scene in several voices with delivery tags the v3 +# model acts on ("[cheerfully] Hello", "[whispering] He is lying"), driven by +# the server through a relayed `dialoguectl` command. Name the cast here — +# "self" always means whatever voice the pet is currently using. +#DIALOGUE=true +#DIALOGUE_MODEL_ID=eleven_v3 +#DIALOGUE_VOICES=narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD + +# ── Self-restart (optional) ───────────────────────────────────────────────── +# `petctl self_restart ` lets Bolt reload the pet after editing its own +# code, so he can check the change live. The code is import-checked first, the +# restart waits for the current turn to finish, and the reason is carried +# across so the new process reports back. The guard refuses more than +# SELF_RESTART_MAX restarts within SELF_RESTART_WINDOW_SECONDS. +#SELF_RESTART=true +#SELF_RESTART_MAX=5 +#SELF_RESTART_WINDOW_SECONDS=900 +# Used instead of ELEVENLABS_MODEL_ID whenever the server picked the voice +# or the reply has non-ASCII in it — the flash_v2 default is English-only. +#ELEVENLABS_MULTILINGUAL_MODEL_ID=eleven_flash_v2_5 + # ── Audio devices (optional — leave blank for the system default) ────────── #MIC_DEVICE= #SPEAKER_DEVICE= diff --git a/CLAUDE.md b/CLAUDE.md index cb21ef6..626a498 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -14,7 +14,8 @@ Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk / click → record utterance → Deepgram STT → + active-window + screen-layout context → POST /desk/converse → [server may relay a shell command to run on this machine, or a `petctl` pseudo-command that moves/emotes the pet, jumps it -to another monitor, or reads a screen's text back instead] → reply → +to another monitor, reads a screen's text back, or plays a multi-voice scene +instead] → reply (optionally tagged with a voice the server picked for it) → ElevenLabs streaming TTS (or offline pyttsx3 fallback) → speakers`, with the pet sprite/speech bubble reflecting state throughout, and playback interruptible by talking over it (barge-in). @@ -79,7 +80,10 @@ logs a missing-config message and exits its thread instead of starting. `/desk/tool_result` until the server sends a final `reply` (capped at `_MAX_RELAY_HOPS`). This is the same "full desktop control" trust model as the server repo's other desk clients — commands only ever originate from - the user's own voice/click requests in their own session. `list_outbox_files` + the user's own voice/click requests in their own session. A final reply is + returned as a `Reply(text, voice_id, voice_name)` rather than a bare string, + because the server can tag it with a voice — see "Voices" below. + `list_outbox_files` / `download_outbox_file` hit the same `/desk/files` and `/desk/files/` endpoints the server's `deliver_files` tool queues onto — see `file_delivery.py`. - **`file_delivery.py`** — the filesystem half of receiving files the server @@ -104,7 +108,11 @@ logs a missing-config message and exits its thread instead of starting. `play_stream()` start playback on the first chunk; `chunks_to_int16()` carries odd bytes across HTTP chunk boundaries, without which everything after the first split sample plays as static — falling back to whole-clip - PCM then offline `pyttsx3`), `barge_in.py` (two detectors behind one + PCM then offline `pyttsx3`; every entry point takes an optional `voice_id` + overriding `ELEVENLABS_VOICE_ID`, and `model_for()` picks the multilingual + model whenever there's an override or non-ASCII text, since the default + `eleven_flash_v2` is English-only and would read either as garbled + phonetic English rather than failing), `barge_in.py` (two detectors behind one `reset()`/`check()` shape, chosen by `BARGE_IN_MODE` via `make_detector`: **wake** (default) scores every frame with the same openWakeWord model the idle listener uses, so only the wake phrase cuts playback; **energy** is the @@ -115,8 +123,8 @@ logs a missing-config message and exits its thread instead of starting. accepts an injectable stream/model/protocol so tests don't need real audio hardware or a display. - **`pet_actions.py`** — `petctl` pseudo-commands (`petctl move top-left`, - `petctl emote wave`, `say`/`wander`/`nap`, plus the screen verbs - `jump`/`monitors`/`read`). The desk API has no "move the pet" payload type + `petctl emote wave`, `say`/`wander`/`nap`, the screen verbs + `jump`/`monitors`/`read`, and `voice reset`). The desk API has no "move the pet" payload type and this repo can't change the server, so these ride the existing shell-command relay: `controller._handle_command` parses them and they never reach `subprocess`; anything else is a real shell command exactly as before. @@ -125,7 +133,51 @@ logs a missing-config message and exits its thread instead of starting. module doesn't have, so the spec passes through to `monitors.resolve()`. Query verbs (`monitors`, `read`) are answered in `_handle_command` rather than by `pet_actions.describe()`, because their output *is* the point: it - goes back up the tool-result relay for Bolt to use in his reply. + goes back up the tool-result relay for Bolt to use in his reply — as is + `voice reset`, which reports what it dropped since the server can't see + which voice is in use. `voice` only ever resets: picking one is the + server's job (`speak_as`, which it already knows how to use), so a + `petctl voice ` attempt is an error pointing back at that marker. +- **`self_restart.py`** — `petctl self_restart`, the pet restarting itself so + Bolt can *see* a code change he just made instead of waiting for a human to + restart it. Three problems shape it, and all three are the interesting part. + (1) The restart can't happen inline: killing the process mid-turn would drop + the HTTP tool relay before the result was posted, leaving the server to wait + out its timeout on a turn that can never finish — so the command only + *arms* it (`controller._arm_self_restart`) and + `controller._maybe_self_restart` fires it after the reply is spoken, the + same "only between turns" rule the updater follows. (2) A broken edit must + not be fatal, so `preflight()` imports the package in a **subprocess** + before arming — this process holds the old modules, so an in-process import + would pass on a file that no longer parses — and a SyntaxError comes back as + the command's output, in the same turn, with the pet still running. (3) The + reason has to outlive the process, so it's written to + `~/.cache/bolt-pet/restart_context.json` (never inside the repo Bolt is + editing) and read on the way back up by `controller._report_self_restart`, + which posts it to the server as an ordinary turn — that's what makes + "restart and check the sprites load" finish as a spoken sentence rather than + a silence. `check_loop_guard` refuses after `SELF_RESTART_MAX` restarts in + `SELF_RESTART_WINDOW_SECONDS`, so an edit-restart-crash cycle stops itself. + Off switch: `SELF_RESTART=false`. +- **`dialogue.py`** — `dialoguectl` pseudo-commands: a multi-voice *scene* + through ElevenLabs' Text to Dialogue endpoint (`audio/tts. + synthesize_dialogue`), checked in `_handle_command` between petctl and + filectl. Same single-line-JSON wire format as filectl and for the same + reason (the server's `command` marker captures only up to the next + newline), and it accepts the ElevenLabs field names (`inputs`/`voice_id`) + as well as its own (`lines`/`voice`) because the model has read that API + and copying its shape is the obvious thing to try. Voices are *named* + (`DIALOGUE_VOICES` maps names to ids) rather than pasted as raw ids, and + `self` resolves to whatever voice the pet is speaking with right now — + including a `speak_as` pick — so Bolt sounds like himself in his own + scenes. The API's limits (10 distinct voices, ~2000 characters) are + enforced *before* the request so a mistake comes back up the tool-result + relay as a sentence Bolt can act on rather than an HTTP 422 he can't see. + Unlike the normal reply path there is no streaming variant, so a scene is + whole-clip: `controller._play_dialogue` plays it with the same bubble, + transcript and barge-in handling a spoken reply gets, and returns to + THINKING afterwards (not IDLE) because the server is still waiting on the + tool result — that leg is why `state.py` allows TALKING -> THINKING. - **`file_ops.py`** — `filectl` pseudo-commands, checked in `_handle_command` right after petctl and before falling through to a real shell command. Executing arbitrary commands already worked via the shell relay @@ -274,9 +326,38 @@ logs a missing-config message and exits its thread instead of starting. name rather than by `PetState`, with `has()` reporting whether a key is backed by real art so callers can decline a placeholder instead of trotting a blob across the desktop; `tray.py` is the system tray menu (talk now / mute / nap / wander / - click-through / history / wake-word tuning / quit) — the pet window has no - title bar or taskbar entry; `history_window.py` and `wake_tuner.py` are the - two dialogs it opens. + click-through / history / wake-word tuning / use-default-voice / quit) — the + pet window has no title bar or taskbar entry; `history_window.py` and + `wake_tuner.py` are the two dialogs it opens. + +### Voices (the server's `speak_as`) + +Ask Bolt to talk like someone else, or in another language, and the *server* +does the picking: its desk-only `voice_search` marker browses the ElevenLabs +voice library, and `speak_as: ` on the final reply tags that reply +with the chosen voice (adding a Voice Library pick to the ElevenLabs account +first, so the id is usable by the time it reaches us). Nothing about that is +this repo's to decide — all the client owes it is actually speaking in the +voice it was handed: `converse()` returns it on `Reply`, `_apply_voice()` +records it, and `_speak()` passes it to `tts.speak(voice_id=...)`. + +Two things are decided *here*, though, because the server can't: + +- **The voice sticks** (`VOICE_STICKY`, default on). The server tags one + reply and strips the marker before storing the turn, so it never sees the + id again — "keep talking like that" would send it searching for a voice all + over again, and it'd likely land on a different one. Holding the id + client-side is what makes the rest of the conversation stay in that voice. + An untagged reply therefore never *changes* the voice; only a new + `speak_as`, `VOICE_STICKY=false`, or a reset does. +- **There's a way back.** Since the server was never told Bolt's own voice + id, it can't ask for it back with `speak_as` — so reverting is local: the + tray's **Use default voice** entry (enabled only while a picked voice is + in use, kept in sync by the `voice_changed` signal), a restart, or + `petctl voice reset`, which is what lets Bolt honour "go back to your + normal voice" out loud. That last one needs the server's pet prompt block + (`ai/desk_api.py`, `pet_tools`) to mention the verb, or the model never + emits it — the desk API's prompt is where petctl is advertised. ### Wake-word detection diff --git a/README.md b/README.md index 9efeee1..6f206ab 100644 --- a/README.md +++ b/README.md @@ -69,8 +69,8 @@ limitations). speaks, and barging in starts your next turn immediately (`BARGE_IN`). - Right-click the tray icon for **Talk now**, **Mute mic**, **Nap**, **Wander around**, **Click through the pet**, **History…**, **Wake word - tuning…** and **Quit** — the pet window itself has no title bar or taskbar - entry. + tuning…**, **Use default voice** and **Quit** — the pet window itself has + no title bar or taskbar entry. - **Click the speech bubble** to copy what it just said; the tray's **History…** window keeps the last `HISTORY_LIMIT` turns. @@ -80,8 +80,8 @@ limitations). listening/thinking/talking or while a bubble is up. - **Moves and emotes on command.** Bolt can relay `petctl move top-left`, `petctl emote wave|hop|spin|nod|shake`, `petctl say ...`, `petctl wander - on|off`, `petctl nap on|off`. These are intercepted here and never reach a - shell. + on|off`, `petctl nap on|off`, `petctl voice reset`, and `dialoguectl` for a + multi-voice scene. These are intercepted here and never reach a shell. - **Naps** during `QUIET_HOURS` (e.g. `23:00-08:00`) or while a fullscreen app is focused (`DND_ON_FULLSCREEN`) — it dims, stops wandering, and makes no proactive noise. It still answers when you speak to it. @@ -90,6 +90,41 @@ limitations). notifications get forwarded to the server, so it can tell you the deploy went green. Off by default: each one costs a round trip. +## Speaking in another voice + +Ask for a different voice — "use a clearer voice", "talk like a pirate", "say +that in Japanese" — and Bolt searches the ElevenLabs voice library on the +server, picks one, and tags his reply with it (`speak_as`); the pet is what +actually speaks in it. A Voice Library pick is added to your ElevenLabs +account automatically the first time it's used, and non-English replies (or +any picked voice) go through `ELEVENLABS_MULTILINGUAL_MODEL_ID` rather than +the English-only `eleven_flash_v2` default. + +The new voice **stays on** for the rest of the conversation, because the +server tags a single reply and doesn't remember which voice it chose — so +"keep talking like that" would otherwise send it hunting for a voice again. +To get his own voice back: ask him ("use your normal voice" — he relays +`petctl voice reset`), use **Use default voice** in the tray menu (greyed +out unless a picked voice is active), or restart the pet. Set +`VOICE_STICKY=false` in `.env` if you'd rather each pick lasted exactly one +reply. + +## Multi-voice dialogue + +Ask for a scene — "do the argument between the two of them", "read that back +as a radio play" — and Bolt can relay a `dialoguectl` command that the pet +renders through ElevenLabs' Text to Dialogue endpoint: several voices in one +take, with delivery tags the v3 model acts on (`[cheerfully]`, `[whispering]`, +`[stuttering]`). One request per scene, so the voices actually react to each +other instead of sounding like clips glued together. + +Name the cast in `.env` (`DIALOGUE_VOICES=narrator:9BWts…,villain:IKne3…`); +the name `self` always means whatever voice the pet is currently using, so +Bolt sounds like himself in his own scenes — including after a `speak_as` +switch. Scenes show up in the speech bubble with the tags stripped, count as +normal speech for the transcript, and can be talked over like any other reply. +`DIALOGUE=false` turns the whole thing off on this device. + ## Wake-word detection `bolt_pet/audio/wake_word.py` feeds every mic frame into `thunderbolt.onnx` diff --git a/bolt_pet/audio/tts.py b/bolt_pet/audio/tts.py index c0b2917..a03cb89 100644 --- a/bolt_pet/audio/tts.py +++ b/bolt_pet/audio/tts.py @@ -5,11 +5,16 @@ desk_client/bolt_desk.py which shells out because it only targets Linux. Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech on macOS, espeak on Linux) if ElevenLabs isn't configured or the request fails, so the pet can still talk with zero cloud config. + +Every entry point takes an optional *voice_id* that overrides +`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as` +reply marker reaches the speakers (see controller._apply_voice). The offline +fallback has no such concept and always sounds like itself. """ from __future__ import annotations -from typing import Iterable, Iterator +from typing import Iterable, Iterator, Optional import numpy as np import requests @@ -21,18 +26,40 @@ class TtsError(Exception): pass -def synthesize_pcm(text: str) -> tuple[np.ndarray, int]: +def voice_for(voice_id: Optional[str] = None) -> str: + """The voice this call should use: an override (server `speak_as`) if + given, else the configured default.""" + return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID + + +def model_for(text: str, voice_id: Optional[str] = None) -> str: + """Which ElevenLabs model to synthesize with. + + The default (`eleven_flash_v2`) is English-only, and both things that + reach this branch mean the reply probably isn't English: a voice the + server picked mid-conversation is nearly always about a language or an + accent, and non-ASCII text can't be English at all. Rendering either one + through the English model gets you a mangled phonetic reading rather + than a failure, which is worse — so those go through the multilingual + model instead.""" + if (voice_id or "").strip() or not text.isascii(): + return config.ELEVENLABS_MULTILINGUAL_MODEL_ID + return config.ELEVENLABS_MODEL_ID + + +def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]: """Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure — callers should fall back to speak_offline() rather than treating this as fatal.""" - if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID): + voice = voice_for(voice_id) + if not (config.ELEVENLABS_API_KEY and voice): raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set") try: response = requests.post( - f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}", + f"https://api.elevenlabs.io/v1/text-to-speech/{voice}", headers={"xi-api-key": config.ELEVENLABS_API_KEY}, params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"}, - json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID}, + json={"text": text, "model_id": model_for(text, voice_id)}, timeout=60, ) response.raise_for_status() @@ -44,20 +71,23 @@ def synthesize_pcm(text: str) -> tuple[np.ndarray, int]: return pcm, config.TTS_SAMPLE_RATE -def stream_pcm(text: str, chunk_bytes: int = 4096) -> Iterator[np.ndarray]: +def stream_pcm( + text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None +) -> Iterator[np.ndarray]: """Same audio as synthesize_pcm(), but yielded as it arrives from ElevenLabs' /stream endpoint so playback can start on the first chunk (~300ms) instead of after the whole clip is synthesized. Raises TtsError before yielding anything if the request itself fails, so callers can fall back cleanly; a mid-stream failure just ends the generator.""" - if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID): + voice = voice_for(voice_id) + if not (config.ELEVENLABS_API_KEY and voice): raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set") try: response = requests.post( - f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}/stream", + f"https://api.elevenlabs.io/v1/text-to-speech/{voice}/stream", headers={"xi-api-key": config.ELEVENLABS_API_KEY}, params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"}, - json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID}, + json={"text": text, "model_id": model_for(text, voice_id)}, timeout=60, stream=True, ) @@ -83,6 +113,55 @@ def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]: yield np.frombuffer(data[:usable], dtype=np.int16) +def synthesize_dialogue( + inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None +) -> tuple[np.ndarray, int]: + """Multi-voice scene via ElevenLabs Text to Dialogue. + + One request, one take: the whole exchange is synthesized together, which + is the point — the model hears the previous line, so reactions and timing + land instead of sounding like separately-rendered clips. + + Same PCM-over-`requests` posture as the rest of this module (no SDK, no + `play()` shelling out to ffplay), so playback is the same sounddevice path + everything else uses and barge-in works on it unchanged. There is no + documented streaming variant, and a scene is a short set piece anyway, so + this is whole-clip only. + """ + if not (config.ELEVENLABS_API_KEY and inputs): + raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)") + body: dict = { + "inputs": [ + {"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")} + for entry in inputs + ], + "model_id": model_id or config.DIALOGUE_MODEL_ID, + } + if stability is not None: + body["settings"] = {"stability": float(stability)} + try: + response = requests.post( + "https://api.elevenlabs.io/v1/text-to-dialogue", + headers={"xi-api-key": config.ELEVENLABS_API_KEY}, + params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"}, + json=body, + timeout=120, # a multi-voice take is slower to render than one line + ) + response.raise_for_status() + except Exception as exc: + detail = "" + # The API explains refusals (character limit, unknown voice) in the + # body; surfacing it is what lets Bolt fix the call and retry. + body_text = getattr(getattr(exc, "response", None), "text", "") + if body_text: + detail = f" — {body_text[:300]}" + raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc + pcm = np.frombuffer(response.content, dtype=np.int16) + if pcm.size == 0: + raise TtsError("ElevenLabs returned no dialogue audio") + return pcm, config.TTS_SAMPLE_RATE + + def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None) -> bool: """Play a whole clip. Returns True if it finished, False if *should_stop* (barge-in) cut it short. *should_stop* is polled while audio plays — each @@ -134,11 +213,12 @@ def speak_offline(text: str) -> None: engine.runAndWait() -def speak(text: str, on_error=None, should_stop=None) -> bool: +def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None) -> bool: """Speak *text*, preferring streaming ElevenLabs, then whole-clip ElevenLabs, then offline TTS. *on_error*, if given, is called with the exception when ElevenLabs fails (useful for logging) — a fallback still runs either way. Returns False if barge-in interrupted playback. + *voice_id* overrides the configured voice for this line only. The text is sanitized first (speech_text.for_speech): server replies are written for a chat window, and a voice reads markdown/emoji literally @@ -149,12 +229,16 @@ def speak(text: str, on_error=None, should_stop=None) -> bool: return True if config.TTS_STREAMING: try: - return play_stream(stream_pcm(text), config.TTS_SAMPLE_RATE, should_stop=should_stop) + return play_stream( + stream_pcm(text, voice_id=voice_id), + config.TTS_SAMPLE_RATE, + should_stop=should_stop, + ) except TtsError as exc: if on_error is not None: on_error(exc) try: - pcm, sample_rate = synthesize_pcm(text) + pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id) return play_pcm(pcm, sample_rate, should_stop=should_stop) except TtsError as exc: if on_error is not None: diff --git a/bolt_pet/config.py b/bolt_pet/config.py index a223b64..ca1d62e 100644 --- a/bolt_pet/config.py +++ b/bolt_pet/config.py @@ -66,9 +66,38 @@ DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3") ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "") ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "") ELEVENLABS_MODEL_ID = os.environ.get("ELEVENLABS_MODEL_ID", "eleven_flash_v2") +# eleven_flash_v2 is English-only, and the two cases that swap the voice +# (server-picked `speak_as`, or a reply with non-ASCII in it) are usually +# exactly the cases where the reply isn't English — see tts.model_for(). +ELEVENLABS_MULTILINGUAL_MODEL_ID = os.environ.get( + "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5" +) # ElevenLabs PCM output formats are named pcm_. TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "24000")) +# Does a voice the server picks (its speak_as marker — "talk like a pirate", +# "say that in Japanese") stay on for later replies, or last one reply only? +# Sticky by default: the server tags a single reply and does *not* keep the +# voice id in its history, so a one-reply-only voice can't be re-used when +# you say "keep talking like that" — it would have to search for a voice +# again. Reset it from the tray ("Use default voice") or by restarting. +VOICE_STICKY = os.environ.get("VOICE_STICKY", "true").lower() in ("1", "true", "yes", "on") + +# ── multi-voice dialogue (ElevenLabs Text to Dialogue) ────────────────────── +# Lets Bolt play a short scene in several voices with delivery tags the v3 +# model acts on ("[cheerfully] Hello"), instead of one voice reading a line. +# Driven by the server through the `dialoguectl` relayed command — see +# dialogue.py. Costs a separate (slower, whole-clip) request per scene, so +# it's a set piece, not the normal reply path. +# +# DIALOGUE_VOICES names the cast: "narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD". +# The name "self" always resolves to the voice the pet is currently using, +# including one the server picked with speak_as. + +DIALOGUE = os.environ.get("DIALOGUE", "true").lower() in ("1", "true", "yes", "on") +DIALOGUE_MODEL_ID = os.environ.get("DIALOGUE_MODEL_ID", "eleven_v3") +DIALOGUE_VOICES = os.environ.get("DIALOGUE_VOICES", "") + # ── mic / VAD (same tuning knobs as bolt_desk.py) ─────────────────────────── MIC_DEVICE = os.environ.get("MIC_DEVICE", "") or None # sounddevice name/index @@ -91,7 +120,7 @@ GRACE_SECONDS = float(os.environ.get("VAD_GRACE_SECONDS", "4")) # keeps feeding it noise. 0 means no cap. FOLLOW_UP_LISTEN = os.environ.get("FOLLOW_UP_LISTEN", "true").lower() in ("1", "true", "yes", "on") -FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "3")) +FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "10")) FOLLOW_UP_GRACE_SECONDS = float(os.environ.get("FOLLOW_UP_GRACE_SECONDS", "7")) COMMAND_TIMEOUT_SECONDS = int(os.environ.get("COMMAND_TIMEOUT_SECONDS", "30")) @@ -110,6 +139,14 @@ SUDO_ASKPASS_HELPER = os.environ.get("SUDO_ASKPASS_HELPER", "") # blank = auto- SUDO_COMMAND_TIMEOUT_SECONDS = int(os.environ.get("SUDO_COMMAND_TIMEOUT_SECONDS", "180")) HEARTBEAT_INTERVAL_SECONDS = float(os.environ.get("HEARTBEAT_INTERVAL_SECONDS", "60")) +# ── self-restart ──────────────────────────────────────────────────────────── +# `petctl self_restart` lets Bolt restart the pet after editing its code, so +# he can see his own change running instead of waiting for someone to restart +# it by hand. The code is import-checked in a subprocess first, and the reason +# is carried across the restart so the new process can report back — see +# self_restart.py. SELF_RESTART_MAX/_WINDOW_SECONDS bound the crash-loop case. +SELF_RESTART = os.environ.get("SELF_RESTART", "true").lower() in ("1", "true", "yes", "on") + # ── barge-in (interrupt playback while the pet is talking) ────────────────── # The mic stays live while the pet talks. BARGE_IN_MODE decides what counts # as an interruption: diff --git a/bolt_pet/controller.py b/bolt_pet/controller.py index c2674df..9a840d3 100644 --- a/bolt_pet/controller.py +++ b/bolt_pet/controller.py @@ -20,10 +20,12 @@ from typing import Optional from PySide6.QtCore import QObject, Signal from . import ( - config, file_delivery, file_ops, history as history_mod, - monitors as monitors_mod, notifications, pet_actions, quiet, - screen_context, screen_text, server_client, speech_text, updater, + config, dialogue as dialogue_mod, file_delivery, file_ops, + history as history_mod, monitors as monitors_mod, notifications, + pet_actions, quiet, screen_context, screen_text, self_restart, + server_client, speech_text, updater, ) +from . import __version__ from .audio import barge_in, mic, stt, tts, wake_word from .state import PetState, PetStateMachine @@ -38,6 +40,7 @@ class PetController(QObject): log = Signal(str) action = Signal(dict) # parsed petctl action for the UI to perform napping = Signal(bool) # quiet hours / fullscreen do-not-disturb + voice_changed = Signal(str) # name of the server-picked voice ("" = default) restart_requested = Signal(str) # version we just updated to finished = Signal() @@ -66,6 +69,13 @@ class PetController(QObject): self._wake_threshold = config.WAKE_WORD_THRESHOLD self._near_misses = wake_word.NearMissLog() + # The voice the server last picked for us with `speak_as` ("" = the + # configured default). Held here rather than passed straight through + # to one tts.speak() call because it's sticky by default — see + # _apply_voice for why. + self._voice_id = "" + self._voice_name = "" + self._barge_in: Optional[barge_in.BargeInDetector] = None self._napping = False self._nap_forced: Optional[bool] = None # petctl nap on/off overrides the schedule @@ -80,6 +90,9 @@ class PetController(QObject): self._last_update_check = 0.0 self._update_pending = False # applied on disk, waiting for the restart + # Armed by `petctl self_restart`, fired after the turn it was asked in + # (see _arm_self_restart for why it can't happen inline). + self._restart_context = None self._notification_watcher: Optional[notifications.NotificationWatcher] = None self._notification_gate = notifications.NotificationGate( @@ -117,6 +130,20 @@ class PetController(QObject): def reset_wake_stats(self) -> None: self._near_misses.clear() + def current_voice(self) -> str: + """Name (or id) of the server-picked voice in use, "" for the default.""" + return self._voice_name or self._voice_id + + def reset_voice(self) -> None: + """Drop a server-picked voice and go back to Bolt's own. The tray's + way out of a voice you didn't want to keep — the server has no way to + ask for the default back, since it never learns what it is.""" + if not self._voice_id: + return + self._voice_id = self._voice_name = "" + self.log.emit("Voice: back to the default.") + self.voice_changed.emit("") + def stop(self) -> None: self._running = False self._talk_now.set() # wake up anything blocked waiting on it @@ -167,6 +194,7 @@ class PetController(QObject): except Exception as exc: self.log.emit(f"Server not reachable yet ({exc}) — will keep trying per-request.") self._start_notification_bridge() + self._report_self_restart() self._loop() if self._notification_watcher is not None: self._notification_watcher.stop() @@ -260,8 +288,10 @@ class PetController(QObject): return self._check_deliveries() - self._speak(reply) + self._apply_voice(reply) + self._speak(reply.text) self._state.transition(PetState.IDLE) + self._maybe_self_restart() def _with_context(self, text: str) -> str: """Everything the server gets alongside what you actually said: the @@ -307,6 +337,18 @@ class PetController(QObject): kind = action["action"] if kind == "monitors": return monitors_mod.describe(self._monitors, self._pet_monitor) + if kind == "self_restart": + return self._arm_self_restart(action.get("reason") or "") + if kind == "voice": + # Answered here, not by describe(): the UI has no part in it, + # and the server needs to hear whether there was anything to + # drop — it can't see which voice we're using. + previous = self.current_voice() + self.reset_voice() + return ( + f"[pet] back to your own voice (was {previous})" if previous + else "[pet] already using your own voice" + ) if kind == "read": return self._read_screen(action["target"]) if kind == "jump": @@ -327,6 +369,14 @@ class PetController(QObject): self.action.emit(action) return pet_actions.describe(action) + try: + scene = dialogue_mod.parse(command) + except dialogue_mod.DialogueError as exc: + self.log.emit(f"dialoguectl: {exc}") + return f"[dialogue] {exc}" + if scene is not None: + return self._play_dialogue(scene) + try: file_action = file_ops.parse(command) except file_ops.FileOpError as exc: @@ -365,6 +415,159 @@ class PetController(QObject): self.log.emit(f"Reading monitor {monitor.number} ({monitor.name})…") return screen_text.read_monitor(monitor, limit) + def _arm_self_restart(self, reason: str) -> str: + """`petctl self_restart` — check the code, then arm a restart. + + Nothing restarts here. The tool result has to get back up the relay + before this process can die (otherwise the server waits out its + timeout on a turn that will never finish), so the restart is armed and + `_maybe_self_restart` fires it once the turn has been spoken. The + preflight import runs *now*, in this turn, so a syntax error Bolt just + introduced comes back as something he can read and fix rather than as + a pet that never comes back.""" + if not config.SELF_RESTART: + return "[pet] self-restart is disabled on this device (SELF_RESTART=false)" + if self._restart_context is not None: + return "[pet] a restart is already armed for the end of this turn" + try: + self_restart.check_loop_guard(self_restart.load()) + self.log.emit("Self-restart requested — checking the code imports first…") + self_restart.preflight() + except self_restart.RestartError as exc: + self.log.emit(f"Self-restart refused: {exc}") + return f"[pet] restart refused — {exc}" + + recent = [entry.text[:120] for entry in self.history.entries()[-4:]] + self._restart_context = self_restart.arm( + reason or "no reason given", + verify=reason, + version=__version__, + session=config.SESSION_ID, + recent=recent, + ) + self.log.emit("Self-restart armed; it happens after this turn.") + return ( + "[pet] code imports cleanly; restarting as soon as this turn finishes. " + "I'll come back and tell you what version I'm on and what I found — " + "wrap up your reply now, the next thing you hear from me is the report." + ) + + def _maybe_self_restart(self) -> bool: + """Fire an armed restart, once the turn is over and the reply spoken. + + Returns True if a restart was requested, so the caller can stop + driving the pipeline — the process is on its way out.""" + if self._restart_context is None: + return False + self._update_pending = True # same latch the updater uses: no double restart + self.log.emit("Restarting now.") + self._state.force(PetState.IDLE) + self.restart_requested.emit(f"self-restart: {self._restart_context.reason[:60]}") + return True + + def _report_self_restart(self) -> None: + """On the way up: tell the server we're back, and why we left. + + Runs once, before the listen loop starts, and only when a context file + was left behind. The report goes through the ordinary conversation + path, so Bolt's answer is spoken out loud like any other turn — which + is what makes "restart and check the sprites load" finish as a + sentence instead of a silence.""" + context = self_restart.load() + if context is None: + return + self_restart.clear() + message = self_restart.report(context, version=__version__) + self.log.emit(f"Back from a self-restart ({context.reason[:80]}).") + self.history.add(history_mod.SYSTEM, message, time.time()) + try: + reply = server_client.converse(message, on_command=self._handle_command) + except server_client.ServerError as exc: + # The restart still worked; only the report failed. Say so locally + # rather than pretending nothing happened. + self.log.emit(f"Couldn't report the restart to the server: {exc}") + return + self._apply_voice(reply) + if reply.text.strip() and not self._napping: + self._speak(reply.text) + self._state.force(PetState.IDLE) + + def _play_dialogue(self, scene: dict) -> str: + """Play a `dialoguectl` scene and report back up the relay. + + This runs *mid-turn* (the server is still waiting on the tool result), + so the pet has to look like it's talking and then go back to waiting — + hence the TALKING → THINKING leg rather than the usual return to IDLE. + Everything a normal reply gets, a scene gets too: the bubble, the + transcript, and barge-in, so a long scene can be talked over exactly + like a long answer.""" + if not config.DIALOGUE: + return "[dialogue] disabled on this device (DIALOGUE=false)" + try: + inputs = dialogue_mod.resolve( + scene, + voices=dialogue_mod.parse_voice_map(config.DIALOGUE_VOICES), + self_voice=self._voice_id or config.ELEVENLABS_VOICE_ID, + ) + except dialogue_mod.DialogueError as exc: + self.log.emit(f"dialoguectl: {exc}") + return f"[dialogue] {exc}" + + text = dialogue_mod.spoken_text(scene) + self.log.emit(f"Dialogue ({len(inputs)} lines): {text[:120]}") + try: + pcm, sample_rate = tts.synthesize_dialogue( + inputs, model_id=scene.get("model"), stability=scene.get("stability") + ) + except tts.TtsError as exc: + self.log.emit(f"Dialogue failed: {exc}") + # Reported, not raised: the server can read this, shorten the + # scene or fix the voice, and try again inside the same turn. + return f"[dialogue] couldn't synthesize it: {exc}" + + resume = self._state.state + self._state.transition(PetState.TALKING) + self.said.emit(speech_text.for_display(text)) + self.history.add(history_mod.PET, text, time.time()) + + should_stop = None + if self._barge_in is not None: + self._barge_in.reset() + should_stop = self._barge_in.check + completed = tts.play_pcm(pcm, sample_rate, should_stop=should_stop) + if self._barge_in is not None: + self._barge_in.reset() # the pet's own voices are in the wake window + if resume in (PetState.THINKING, PetState.IDLE): + self._state.transition(resume) + + if not completed: + return dialogue_mod.describe(scene) + " (interrupted — they talked over it)" + return dialogue_mod.describe(scene) + + def _apply_voice(self, reply) -> None: + """Adopt (or drop) the voice the server tagged this reply with. + + The server's `speak_as` marker names an ElevenLabs voice it just + picked — and it adds a Voice Library pick to the account first, so by + the time the id gets here it's usable for TTS. It tags *one* reply, + but the voice sticks by default: the server strips the marker before + storing the turn, so it can't recall the id later, and "keep talking + like that" would otherwise send it searching for a voice all over + again. `VOICE_STICKY=false` makes each pick last exactly one reply. + + Untagged replies never *change* the voice — with stickiness on they + just keep whatever's in use, which is what makes the rest of the + conversation stay in the requested voice.""" + voice_id = getattr(reply, "voice_id", "") + if voice_id: + if voice_id != self._voice_id: + self._voice_id = voice_id + self._voice_name = getattr(reply, "voice_name", "") or "" + self.log.emit(f"Voice: {self.current_voice()}") + self.voice_changed.emit(self.current_voice()) + elif not config.VOICE_STICKY: + self.reset_voice() + def _speak(self, text: str) -> None: self._state.transition(PetState.TALKING) # Bubble gets the markdown stripped but emoji kept (it can't render @@ -382,6 +585,7 @@ class PetController(QObject): text, on_error=lambda exc: self.log.emit(f"TTS failed: {exc}"), should_stop=should_stop, + voice_id=self._voice_id or None, ) # Read the scoring history *before* resetting, or the log reports the # blank counters instead of what actually fired. @@ -504,8 +708,9 @@ class PetController(QObject): self.log.emit(f"Couldn't forward notification: {exc}") return self._check_deliveries() - if reply.strip(): - self._speak(reply) + self._apply_voice(reply) + if reply.text.strip(): + self._speak(reply.text) self._state.transition(PetState.IDLE) # ── file delivery ──────────────────────────────────────────────────── diff --git a/bolt_pet/dialogue.py b/bolt_pet/dialogue.py new file mode 100644 index 0000000..a14d135 --- /dev/null +++ b/bolt_pet/dialogue.py @@ -0,0 +1,219 @@ +"""`dialoguectl` — multi-voice dialogue playback (ElevenLabs Text to Dialogue). + +Normal replies are one voice saying one thing (audio/tts.py). This is the +other mode: a short *scene* — two or more voices, with delivery tags the v3 +model acts on (`[cheerfully]`, `[stuttering]`, `[whispering]`) — synthesized +as a single take so the timing and reactions between lines actually sound +like a conversation rather than clips glued together. + +Wire format, the same discipline as file_ops.py and for the same reason: it +rides the server's ordinary `command` tool marker, whose extractor only +captures up to the next newline, so the payload is a **single-line compact +JSON object**. + + dialoguectl {"lines": [{"voice": "self", "text": "[cheerfully] Morning!"}, + {"voice": "narrator", "text": "[whispering] He lies."}]} + +The ElevenLabs field names are accepted too (`inputs` / `voice_id`), because +the model has read that API and copying its shape is the obvious thing to +try: + + dialoguectl {"inputs": [{"voice_id": "9BWtsMINqrJLrRacOk9x", "text": "hi"}]} + +Voices are *named*, not pasted as ids. `DIALOGUE_VOICES` in .env maps names +to ids (`narrator:9BWts…,villain:IKne3…`), and `self` always means the voice +the pet is speaking with right now — including a voice the server picked +mid-conversation with `speak_as`, so a scene featuring Bolt sounds like +whoever Bolt currently is. + +Pure parsing and validation here; the HTTP call is +`audio/tts.synthesize_dialogue` and the playback/state handling is +`controller._play_dialogue`, matching the parse/execute split used by +pet_actions.py and file_ops.py. + +The API's own limits are enforced *here*, before the request goes out, so a +mistake comes back through the tool-result relay as a sentence Bolt can act +on ("too many characters, split it") rather than as an HTTP 422 he can't see. +""" + +from __future__ import annotations + +import json +import re +from typing import Iterable, Optional + +_PREFIXES = ("dialoguectl", "dialogue", "scene") + +# ElevenLabs Text to Dialogue limits (docs, 2026-07): at most 10 distinct +# voice ids per request and ~2000 characters across all inputs. +MAX_VOICES = 10 +MAX_CHARS = 2000 + +# Names that always mean "the voice the pet is using right now". +SELF_NAMES = ("self", "bolt", "me", "pet") + +# A raw ElevenLabs voice id: 20 URL-safe characters, no separators. Used to +# tell "the model pasted an id" from "the model used a name". +_VOICE_ID_RE = re.compile(r"^[A-Za-z0-9]{20}$") + + +class DialogueError(Exception): + """Bad dialoguectl syntax or an unusable request — reported back to the + server as this command's output.""" + + +def is_dialogue_command(command: str) -> bool: + parts = (command or "").strip().split(None, 1) + return bool(parts) and parts[0].lower() in _PREFIXES + + +def parse(command: str) -> Optional[dict]: + """Parse `dialoguectl ` into {"lines": [{"voice", "text"}], ...}. + + Returns None if this isn't a dialogue command at all (the caller then + tries filectl, then a real shell command). Raises DialogueError on a + dialogue command that doesn't make sense.""" + if not is_dialogue_command(command): + return None + _, _, payload = (command or "").strip().partition(" ") + payload = payload.strip() + if not payload: + raise DialogueError( + 'dialoguectl needs a JSON argument, e.g. dialoguectl {"lines": ' + '[{"voice": "self", "text": "[cheerfully] hello"}]}' + ) + try: + data = json.loads(payload) + except json.JSONDecodeError as exc: + raise DialogueError( + f"couldn't parse the JSON ({exc}). It must be one line of compact " + "JSON — put line breaks inside text as \\n, never as real newlines." + ) from exc + if not isinstance(data, dict): + raise DialogueError("the argument must be a JSON object, not a list or a bare value") + + raw_lines = data.get("lines") + if raw_lines is None: + raw_lines = data.get("inputs") # the ElevenLabs field name + if not isinstance(raw_lines, list) or not raw_lines: + raise DialogueError('needs a non-empty "lines" array of {"voice", "text"} objects') + + lines: list[dict] = [] + for index, entry in enumerate(raw_lines, start=1): + if not isinstance(entry, dict): + raise DialogueError(f"line {index} must be an object with 'voice' and 'text'") + text = str(entry.get("text") or "").strip() + if not text: + raise DialogueError(f"line {index} has no text") + voice = str(entry.get("voice") or entry.get("voice_id") or "self").strip() + lines.append({"voice": voice, "text": text}) + + action = {"action": "dialogue", "lines": lines} + model = str(data.get("model") or data.get("model_id") or "").strip() + if model: + action["model"] = model + stability = data.get("stability") + if stability is not None: + try: + action["stability"] = min(1.0, max(0.0, float(stability))) + except (TypeError, ValueError): + raise DialogueError("stability must be a number between 0 and 1") from None + return action + + +def parse_voice_map(spec: str) -> dict[str, str]: + """Parse DIALOGUE_VOICES ("narrator:9BWts…, villain:IKne3…") into a map. + + Malformed entries are skipped rather than raising: a typo in .env should + cost that one voice, not the whole feature.""" + voices: dict[str, str] = {} + for chunk in str(spec or "").split(","): + name, separator, voice_id = chunk.partition(":") + name, voice_id = name.strip().lower(), voice_id.strip() + if separator and name and voice_id: + voices[name] = voice_id + return voices + + +def resolve( + action: dict, + *, + voices: Optional[dict] = None, + self_voice: str = "", +) -> list[dict]: + """Turn parsed lines into the API's `inputs`, resolving names to ids. + + *self_voice* is the pet's current voice (which may be a `speak_as` pick, + not the configured default), so "self" tracks whoever Bolt sounds like + right now.""" + known = dict(voices or {}) + resolved: list[dict] = [] + for index, line in enumerate(action.get("lines") or [], start=1): + name = str(line.get("voice") or "self") + key = name.lower() + if key in SELF_NAMES: + voice_id = self_voice + if not voice_id: + raise DialogueError( + "no voice is configured for the pet itself — set " + "ELEVENLABS_VOICE_ID, or name a voice from DIALOGUE_VOICES" + ) + elif key in known: + voice_id = known[key] + elif _VOICE_ID_RE.match(name): + voice_id = name # a raw id pasted straight from the voice library + else: + available = ", ".join(sorted(known) + list(SELF_NAMES[:1])) or "self" + raise DialogueError( + f"line {index}: unknown voice {name!r}. Known names: {available}. " + "Use one of those, 'self' for your own voice, or a raw voice id." + ) + resolved.append({"text": str(line.get("text") or ""), "voice_id": voice_id}) + + check_limits(resolved) + return resolved + + +def check_limits(inputs: Iterable[dict], *, max_voices: int = MAX_VOICES, + max_chars: int = MAX_CHARS) -> None: + """Enforce the API's own limits before spending a request on a 422.""" + entries = list(inputs) + if not entries: + raise DialogueError("no lines to speak") + distinct = {entry["voice_id"] for entry in entries} + if len(distinct) > max_voices: + raise DialogueError( + f"{len(distinct)} different voices — the limit is {max_voices} per scene" + ) + total = sum(len(entry["text"]) for entry in entries) + if total > max_chars: + raise DialogueError( + f"{total} characters — the limit is {max_chars} per scene. " + "Split it into two dialoguectl calls." + ) + + +def spoken_text(action: dict) -> str: + """The scene as readable text, for the speech bubble and the transcript. + + Delivery tags are stripped: `[cheerfully]` is a stage direction for the + model, not something to show (or, via tts.speak's sanitizer, to read out).""" + parts = [] + for line in action.get("lines") or []: + text = re.sub(r"\[[^\]]{1,40}\]", " ", str(line.get("text") or "")) + text = " ".join(text.split()) + if text: + parts.append(text) + return " ".join(parts) + + +def describe(action: dict, *, played: bool = True) -> str: + """The tool-result string handed back to the server.""" + lines = action.get("lines") or [] + voices = sorted({str(line.get("voice") or "self") for line in lines}) + if not played: + return f"[dialogue] not played ({len(lines)} lines)" + return ( + f"[dialogue] played {len(lines)} line{'s' if len(lines) != 1 else ''} " + f"in {len(voices)} voice{'s' if len(voices) != 1 else ''}: {', '.join(voices)}" + ) diff --git a/bolt_pet/pet_actions.py b/bolt_pet/pet_actions.py index c9376b9..dba41e2 100644 --- a/bolt_pet/pet_actions.py +++ b/bolt_pet/pet_actions.py @@ -44,9 +44,17 @@ HELP = ( "petctl emote <" + "|".join(EMOTES) + ">\n" "petctl say \n" "petctl wander on|off\n" - "petctl nap on|off" + "petctl nap on|off\n" + "petctl voice reset\n" + "petctl self_restart [why]" ) +# `petctl voice` only ever goes one way: back to the configured voice. Picking +# a *different* one is the server's job (its speak_as reply marker), and it +# already knows how — what it has no way to say is "never mind, be yourself +# again", because it was never told which voice that is. +VOICE_RESETS = ("reset", "default", "normal", "own", "back", "mine", "yours") + class ActionError(Exception): """Bad petctl syntax — reported back to the server as command output.""" @@ -129,6 +137,22 @@ def parse(command: str) -> Optional[dict]: raise ActionError("wander needs on or off") return {"action": "wander", "enabled": _bool_arg(args[0])} + if verb == "voice": + target = (args[0].lower() if args else "reset").lstrip("-") + if target not in VOICE_RESETS: + raise ActionError( + f"can't set a voice from petctl (got {args[0]!r}); " + "use the speak_as reply marker to pick one. " + "petctl voice reset goes back to the default voice." + ) + return {"action": "voice", "voice": "default"} + + if verb in ("self_restart", "restart", "reboot"): + # Free text, not a fixed grammar: the argument is a note to the pet's + # *next* process about why it died and what to look at when it comes + # back, so anything the model wants to tell future-itself is valid. + return {"action": "self_restart", "reason": " ".join(args).strip()} + if verb in ("nap", "sleep", "dnd"): if not args: raise ActionError("nap needs on or off") diff --git a/bolt_pet/self_restart.py b/bolt_pet/self_restart.py new file mode 100644 index 0000000..506e017 --- /dev/null +++ b/bolt_pet/self_restart.py @@ -0,0 +1,235 @@ +"""`petctl self_restart` — the pet restarting itself, and remembering why. + +Bolt can already edit this repo through `filectl` and run commands through the +shell relay, which means he can change the pet's own code. What he could not +do is *see the result*: the running process keeps the old modules in memory, +so an edit is invisible until somebody restarts the pet by hand, and by then +the conversation that motivated it is over. That makes the edit-test-review +loop a human errand. + +This closes the loop. The tricky part is that the thing being asked to report +back is the thing that dies, so the mechanism is built around three problems: + +1. **The turn must survive.** A restart mid-turn would kill the HTTP tool + relay before the result was posted, and the server would sit waiting until + it timed out — the conversation lost, with no explanation. So the command + only *arms* the restart: it returns immediately, the turn finishes and Bolt + speaks his reply, and the restart happens after (see + `controller._maybe_self_restart`), exactly like the updater's "only between + turns" rule. +2. **A broken edit must not be fatal.** Before anything is armed, the new code + is imported in a *subprocess* (`preflight`) — this process still holds the + old modules, so importing here would prove nothing. A syntax error comes + back as the command's output, in the same turn, and nothing restarts. That + is the difference between "Bolt broke the pet and lost his own way to fix + it" and "Bolt got a traceback and tried again". +3. **The reason must outlive the process.** The context (why, what to check, + which version, when) is written to disk before exec and read on the way + back up, so the new process can open with "I'm back — you asked me to check + X" instead of amnesia. That report goes to the server as a normal turn, so + Bolt sees the result of his own change and can carry on. + +A loop guard bounds the worst case: `MAX_RESTARTS` inside `WINDOW_SECONDS` +and further self-restarts are refused with a reason, so an edit-restart-crash +cycle stops on its own rather than spinning the process forever. + +Pure-ish and injectable throughout (paths, clock, subprocess runner) so the +whole thing is testable without ever restarting anything. +""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys +import tempfile +import time +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, Callable, Optional + +from . import config + +# Lives in the cache dir, not the repo: it is transient state about *this* +# machine's process, and it must never end up in a git diff of the checkout +# Bolt is editing. +DEFAULT_STATE_PATH = Path.home() / ".cache" / "bolt-pet" / "restart_context.json" + +# Loop guard. Deliberately small: a healthy edit-check cycle is one restart +# per change, and anything hammering past this is a crash loop, not work. +MAX_RESTARTS = int(os.environ.get("SELF_RESTART_MAX", "5")) +WINDOW_SECONDS = float(os.environ.get("SELF_RESTART_WINDOW_SECONDS", "900")) + +# What the preflight subprocess imports. `ui.app` pulls in the widest slice of +# the package (Qt, controller, audio, every helper), so if this imports, a +# restart will at least reach the event loop. +_PREFLIGHT_IMPORT = "import bolt_pet, bolt_pet.controller, bolt_pet.ui.app" + + +class RestartError(Exception): + """A refused restart — reported back to the server as command output.""" + + +@dataclass +class RestartContext: + """What the dying process wants the next one to know.""" + + reason: str = "" + verify: str = "" + armed_at: float = 0.0 + version: str = "" + session: str = "" + recent: list = field(default_factory=list) + restarts: list = field(default_factory=list) # timestamps, for the loop guard + + def as_dict(self) -> dict[str, Any]: + return asdict(self) + + +def _now() -> float: + return time.time() + + +def load(path: Optional[Path] = None) -> Optional[RestartContext]: + """Read the context left by a previous process, or None.""" + target = Path(path or DEFAULT_STATE_PATH) + try: + data = json.loads(target.read_text(encoding="utf-8")) + except (FileNotFoundError, json.JSONDecodeError, OSError): + return None + if not isinstance(data, dict): + return None + known = {field_name for field_name in RestartContext().as_dict()} + return RestartContext(**{k: v for k, v in data.items() if k in known}) + + +def save(context: RestartContext, path: Optional[Path] = None) -> None: + """Persist the context atomically — a half-written file on the way out + would make the next process start confused instead of oriented.""" + target = Path(path or DEFAULT_STATE_PATH) + target.parent.mkdir(parents=True, exist_ok=True) + descriptor, temp_path = tempfile.mkstemp(dir=target.parent, prefix=".restart_", suffix=".tmp") + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + json.dump(context.as_dict(), handle, ensure_ascii=False, indent=1) + os.replace(temp_path, target) + except BaseException: + try: + os.unlink(temp_path) + except OSError: + pass + raise + + +def clear(path: Optional[Path] = None) -> None: + """Consume the context. Called once it has been reported, so the pet + doesn't announce the same restart every time it starts.""" + try: + Path(path or DEFAULT_STATE_PATH).unlink() + except (FileNotFoundError, OSError): + pass + + +def recent_restarts(context: Optional[RestartContext], *, now: Optional[float] = None) -> list: + current = now if now is not None else _now() + stamps = list((context.restarts if context else []) or []) + return [stamp for stamp in stamps if current - float(stamp) <= WINDOW_SECONDS] + + +def check_loop_guard(context: Optional[RestartContext], *, now: Optional[float] = None) -> None: + """Refuse to restart if we've already done it too many times recently.""" + stamps = recent_restarts(context, now=now) + if len(stamps) >= MAX_RESTARTS: + raise RestartError( + f"refusing: {len(stamps)} self-restarts in the last " + f"{int(WINDOW_SECONDS / 60)} minutes. Something is looping — fix the " + "cause, or wait for the window to clear before trying again." + ) + + +def preflight( + repo: Optional[Path] = None, + run: Optional[Callable[..., Any]] = None, + timeout: float = 120.0, +) -> None: + """Import the current source in a subprocess; raise if it's broken. + + This process has the *old* modules loaded, so importing in-process would + happily succeed on a file that no longer parses. Mirrors + `updater._smoke_test`, and exists for the same reason: never hand the + session to code that can't start.""" + runner = run or subprocess.run + root = Path(repo or config.HERE) + try: + completed = runner( + [sys.executable, "-c", _PREFLIGHT_IMPORT], + cwd=str(root), capture_output=True, text=True, timeout=timeout, + env={**os.environ, "QT_QPA_PLATFORM": "offscreen"}, # no display needed to import + ) + except Exception as exc: # subprocess itself failed to run + raise RestartError(f"couldn't run the preflight import check: {exc}") from exc + if completed.returncode != 0: + detail = (completed.stderr or completed.stdout or "").strip() + raise RestartError( + "the current code does not import, so restarting would leave you with " + f"nothing running. Fix this first:\n{detail[-800:]}" + ) + + +def arm( + reason: str, + *, + verify: str = "", + version: str = "", + session: str = "", + recent: Optional[list] = None, + path: Optional[Path] = None, + now: Optional[float] = None, +) -> RestartContext: + """Record why we're about to die, carrying the restart history forward.""" + current = now if now is not None else _now() + previous = load(path) + context = RestartContext( + reason=" ".join(str(reason or "").split())[:400], + verify=" ".join(str(verify or "").split())[:400], + armed_at=current, + version=str(version or ""), + session=str(session or ""), + recent=list(recent or [])[-6:], + restarts=recent_restarts(previous, now=current) + [current], + ) + save(context, path) + return context + + +def report( + context: RestartContext, + *, + version: str = "", + now: Optional[float] = None, +) -> str: + """The message the new process sends the server on the way up. + + Phrased as Bolt reporting to himself, because that is what it is: the + server sees it as an ordinary turn, and the reply comes back through the + normal pipeline — which is what lets "restart and check X" finish as a + sentence spoken out loud.""" + current = now if now is not None else _now() + took = max(0.0, current - float(context.armed_at or current)) + lines = [ + "[pet self-restart] I restarted myself and I'm back up.", + f"- reason: {context.reason or 'not recorded'}", + f"- took: {took:.1f}s", + f"- version now running: {version or 'unknown'}" + + (f" (was {context.version})" if context.version and context.version != version else ""), + ] + if context.verify: + lines.append(f"- you wanted to check: {context.verify}") + if context.recent: + lines.append("- what we were doing before: " + " | ".join(str(x)[:120] for x in context.recent)) + lines.append( + "The new code is loaded and running. If you wanted to verify something, " + "check it now (filectl to read, command to test) and tell the user what you found." + ) + return "\n".join(lines) diff --git a/bolt_pet/server_client.py b/bolt_pet/server_client.py index d9b91b1..f0a60e1 100644 --- a/bolt_pet/server_client.py +++ b/bolt_pet/server_client.py @@ -9,6 +9,11 @@ memory, tools, and persona as Discord chat and the Linux voice client: ... -> POST /desk/tool_result (repeat until the server sends a reply) reply <- returned to caller +A reply can also carry a voice (`voice_id`/`voice_name`), which is how the +server's `speak_as` marker reaches us: Bolt searched the ElevenLabs voice +library, picked one, and tagged the reply with it — the client is what +actually speaks in it. See `Reply` and controller._apply_voice. + Kept dependency-free beyond `requests` so it's easy to unit test with mocks. """ @@ -16,7 +21,7 @@ from __future__ import annotations import subprocess from pathlib import Path -from typing import Callable, Optional +from typing import Callable, NamedTuple, Optional import requests @@ -29,6 +34,17 @@ class ServerError(Exception): """Raised when the server responds with an error payload or unreachable.""" +class Reply(NamedTuple): + """One final reply from the desk API. *voice_id* is set only when the + server tagged this reply with a `speak_as` voice; *voice_name* is the + human-readable name that came with it (may be empty even when the id + isn't). Both empty means "say it in the usual voice".""" + + text: str + voice_id: str = "" + voice_name: str = "" + + def _headers() -> dict: return {"X-Desk-Api-Key": config.API_KEY} @@ -74,7 +90,7 @@ def converse( text: str, on_command: Callable[[str], str] = run_local_command, timeout: float = 120.0, -) -> str: +) -> Reply: """Send one turn of conversation to the desk API, relaying any commands the server sends back until it produces a final reply. @@ -111,7 +127,11 @@ def converse( raise ServerError(f"couldn't reach the server during tool relay: {exc}") from exc if payload.get("type") == "reply": - return str(payload.get("text") or "") + return Reply( + text=str(payload.get("text") or ""), + voice_id=str(payload.get("voice_id") or ""), + voice_name=str(payload.get("voice_name") or ""), + ) raise ServerError(str(payload.get("error") or "unknown server response")) diff --git a/bolt_pet/state.py b/bolt_pet/state.py index 35ba9b4..7747baf 100644 --- a/bolt_pet/state.py +++ b/bolt_pet/state.py @@ -26,11 +26,16 @@ class PetState(str, Enum): # make the pet speak unprompted — a reminder firing, a nudge from the server # — without the user having said anything first, so there's no preceding # LISTENING/THINKING leg for that turn. +# +# TALKING -> THINKING is the mirror case: a `dialoguectl` scene is played +# *mid-turn*, while the server is still waiting on the tool result, so the pet +# talks and then goes back to waiting rather than falling to IDLE (which would +# make it look like the turn had ended). _TRANSITIONS: dict[PetState, set[PetState]] = { PetState.IDLE: {PetState.LISTENING, PetState.TALKING, PetState.ERROR}, PetState.LISTENING: {PetState.THINKING, PetState.IDLE, PetState.ERROR}, PetState.THINKING: {PetState.TALKING, PetState.IDLE, PetState.ERROR}, - PetState.TALKING: {PetState.IDLE, PetState.ERROR}, + PetState.TALKING: {PetState.IDLE, PetState.THINKING, PetState.ERROR}, PetState.ERROR: {PetState.IDLE}, } diff --git a/bolt_pet/ui/app.py b/bolt_pet/ui/app.py index 694cb71..b575c16 100644 --- a/bolt_pet/ui/app.py +++ b/bolt_pet/ui/app.py @@ -80,6 +80,7 @@ def run() -> int: on_set_nap=_set_nap, on_show_history=history_window.show_refreshed, on_show_wake_tuner=tuner_window.show_refreshed, + on_reset_voice=controller.reset_voice, ) def _handle_napping(napping: bool) -> None: @@ -87,6 +88,9 @@ def run() -> int: tray.set_napping(napping) controller.napping.connect(_handle_napping) + # The server can hand Bolt a different voice mid-conversation (speak_as); + # the tray is where you get his own back. + controller.voice_changed.connect(tray.set_voice) # Push-to-talk: a global hook, because the pet window never has focus. # request_talk_now() only sets a threading.Event, so it's safe to call diff --git a/bolt_pet/ui/tray.py b/bolt_pet/ui/tray.py index 9bfcd13..e2d2867 100644 --- a/bolt_pet/ui/tray.py +++ b/bolt_pet/ui/tray.py @@ -1,6 +1,7 @@ """System tray icon — the pet window is frameless with no taskbar entry, so this menu is the only always-available way to control or exit it: talk now, -mute, wander, click-through, nap, history, wake-word tuning, quit. +mute, wander, click-through, nap, history, wake-word tuning, voice reset, +quit. Every entry is a plain callback passed in by ui/app.py; this file knows nothing about the controller or the pet window. @@ -46,6 +47,7 @@ class PetTray(QSystemTrayIcon): on_set_nap: Optional[Callable[[bool], None]] = None, on_show_history: Optional[Callable[[], None]] = None, on_show_wake_tuner: Optional[Callable[[], None]] = None, + on_reset_voice: Optional[Callable[[], None]] = None, parent=None, ): super().__init__(_make_icon(muted=False), parent) @@ -102,6 +104,16 @@ class PetTray(QSystemTrayIcon): tuner_action.triggered.connect(on_show_wake_tuner) menu.addAction(tuner_action) + # Only ever enabled while a server-picked voice (speak_as) is in use — + # it's the way back from "talk like a pirate", which nothing else + # undoes short of a restart. + self._voice_action = None + if on_reset_voice is not None: + self._voice_action = QAction("Use default voice", menu) + self._voice_action.setEnabled(False) + self._voice_action.triggered.connect(on_reset_voice) + menu.addAction(self._voice_action) + menu.addSeparator() quit_action = QAction("Quit", menu) quit_action.triggered.connect(on_quit) @@ -115,6 +127,16 @@ class PetTray(QSystemTrayIcon): self._mute_action.setChecked(self._muted) self._refresh_icon() + def set_voice(self, voice: str) -> None: + """Reflect the voice the controller is speaking in — a name (or id) + when the server picked one, "" for Bolt's own.""" + if self._voice_action is None: + return + self._voice_action.setEnabled(bool(voice)) + self._voice_action.setText( + f"Use default voice (now: {voice})" if voice else "Use default voice" + ) + def set_napping(self, napping: bool) -> None: """Reflect a nap the *controller* decided on (quiet hours, fullscreen, or a petctl command) — not just ones clicked here.""" diff --git a/tests/test_controller.py b/tests/test_controller.py index db5e782..52da051 100644 --- a/tests/test_controller.py +++ b/tests/test_controller.py @@ -41,10 +41,10 @@ def test_full_turn_happy_path(monkeypatch, ctrl): monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16)) monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's the weather") - monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: "sunny and 72") + monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: controller_mod.server_client.Reply("sunny and 72")) spoken = [] monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1]) + lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1]) ctrl._handle_conversation_turn() @@ -150,7 +150,7 @@ def test_heartbeat_speaks_a_pending_announcement_when_idle(monkeypatch, ctrl): monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: "don't forget your 3pm") spoken = [] monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1]) + lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1]) said = _capture(ctrl.said) ctrl._maybe_heartbeat() diff --git a/tests/test_controller_features.py b/tests/test_controller_features.py index ed7e9d7..f76c674 100644 --- a/tests/test_controller_features.py +++ b/tests/test_controller_features.py @@ -15,6 +15,7 @@ from PySide6.QtWidgets import QApplication from bolt_pet import controller as controller_mod from bolt_pet.notifications import Notification +from bolt_pet.server_client import Reply from bolt_pet.state import PetState sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) @@ -159,7 +160,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch ctrl._barge_in = detector logs = _capture(ctrl.log) - def interrupted_playback(text, on_error=None, should_stop=None): + def interrupted_playback(text, on_error=None, should_stop=None, voice_id=None): # What really happens: frames get scored during playback, then one # clears the threshold and playback aborts. detector._frames, detector._peak, detector._last = 7, 0.81, 0.81 @@ -179,7 +180,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl): logs = _capture(ctrl.log) monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: False) # interrupted + lambda text, on_error=None, should_stop=None, voice_id=None: False) # interrupted ctrl._speak("a very long explanation") @@ -189,7 +190,7 @@ def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl): def test_uninterrupted_playback_does_not_queue_a_turn(monkeypatch, ctrl): monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: True) + lambda text, on_error=None, should_stop=None, voice_id=None: True) ctrl._speak("short answer") assert not ctrl._talk_now.is_set() @@ -201,7 +202,7 @@ def spoke(monkeypatch): """Playback that always completes, so only the follow-up rule decides whether another turn is queued.""" monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: True) + lambda text, on_error=None, should_stop=None, voice_id=None: True) def test_a_reply_ending_in_a_question_keeps_listening(spoke, ctrl): @@ -289,7 +290,7 @@ def test_follow_up_can_be_turned_off(spoke, monkeypatch, ctrl): def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl): monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: False) + lambda text, on_error=None, should_stop=None, voice_id=None: False) ctrl._follow_ups = 3 ctrl._speak("a very long explanation") @@ -300,7 +301,7 @@ def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl): def test_speech_is_recorded_in_the_history(monkeypatch, ctrl): monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: True) + lambda text, on_error=None, should_stop=None, voice_id=None: True) ctrl._speak("**bold** reply") assert ctrl.history.last().text == "**bold** reply" # raw, for copy/paste @@ -314,10 +315,10 @@ def test_the_active_window_rides_along_with_the_utterance(monkeypatch, ctrl): monkeypatch.setattr(controller_mod.screen_context, "context_for", lambda text: f"{text}\n\n[on screen right now: app.py]") monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: True) + lambda text, on_error=None, should_stop=None, voice_id=None: True) sent = [] monkeypatch.setattr(controller_mod.server_client, "converse", - lambda text, on_command=None: sent.append(text) or "that's a KeyError") + lambda text, on_command=None: sent.append(text) or Reply("that's a KeyError")) ctrl._handle_conversation_turn() @@ -370,10 +371,10 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl): lambda *a, **k: np.zeros(10, dtype=np.int16)) monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "you awake?") monkeypatch.setattr(controller_mod.server_client, "converse", - lambda text, on_command=None: "always") + lambda text, on_command=None: Reply("always")) spoken = [] monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: spoken.append(text) or True) + lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True) ctrl.set_napping(True) ctrl._handle_conversation_turn() @@ -388,9 +389,9 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl): ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0) sent, spoken = [], [] monkeypatch.setattr(controller_mod.server_client, "converse", - lambda text, on_command=None: sent.append(text) or "your build is green") + lambda text, on_command=None: sent.append(text) or Reply("your build is green")) monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: spoken.append(text) or True) + lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True) ctrl._queue_notification(Notification(app="CI", summary="Build finished", body="")) ctrl._drain_notifications() @@ -506,9 +507,9 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_ lambda *a, **k: np.zeros(10, dtype=np.int16)) monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "send me that file") monkeypatch.setattr(controller_mod.server_client, "converse", - lambda text, on_command=None: "it's on the way") + lambda text, on_command=None: Reply("it's on the way")) monkeypatch.setattr(controller_mod.tts, "speak", - lambda text, on_error=None, should_stop=None: True) + lambda text, on_error=None, should_stop=None, voice_id=None: True) monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path) monkeypatch.setattr(controller_mod.server_client, "list_outbox_files", lambda: [{"id": "abc", "name": "notes.txt", "size": 2}]) @@ -518,3 +519,294 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_ ctrl._handle_conversation_turn() assert (tmp_path / "notes.txt").read_bytes() == b"hi" + + +# ── server-picked voice (the desk API's speak_as marker) ──────────────────── + +def _voice_turn(monkeypatch, ctrl, reply, said="talk like a pirate"): + """Run one full conversation turn whose reply is *reply*, returning the + voice_id each tts.speak() call was given.""" + voices = [] + monkeypatch.setattr(controller_mod.mic, "record_utterance", + lambda *a, **k: np.zeros(10, dtype=np.int16)) + monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said) + monkeypatch.setattr(controller_mod.server_client, "converse", + lambda text, on_command=None: reply) + monkeypatch.setattr( + controller_mod.tts, "speak", + lambda text, on_error=None, should_stop=None, voice_id=None: + voices.append(voice_id) or True, + ) + ctrl._handle_conversation_turn() + return voices + + +def test_a_speak_as_reply_is_spoken_in_that_voice(monkeypatch, ctrl): + voices = _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence")) + assert voices == ["VOICE1"] + assert ctrl.current_voice() == "Terence" + + +def test_the_picked_voice_sticks_for_later_replies(monkeypatch, ctrl): + """The server tags one reply and doesn't keep the id in its history, so + it can't re-request the voice when you say "keep talking like that".""" + monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", True) + _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence")) + voices = _voice_turn(monkeypatch, ctrl, Reply("Still me."), said="and now?") + assert voices == ["VOICE1"] + + +def test_voice_stickiness_can_be_turned_off(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", False) + _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence")) + voices = _voice_turn(monkeypatch, ctrl, Reply("Back to normal."), said="and now?") + assert voices == [None] + assert ctrl.current_voice() == "" + + +def test_a_new_pick_replaces_the_old_one(monkeypatch, ctrl): + changes = _capture(ctrl.voice_changed) + _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence")) + voices = _voice_turn(monkeypatch, ctrl, Reply("こんにちは。", "VOICE2", "Asahi"), + said="say that in Japanese") + assert voices == ["VOICE2"] + assert changes == ["Terence", "Asahi"] + + +def test_resetting_the_voice_goes_back_to_the_default(monkeypatch, ctrl): + _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence")) + changes = _capture(ctrl.voice_changed) + + ctrl.reset_voice() + + assert ctrl.current_voice() == "" + assert changes == [""] # the tray's menu entry follows this signal + voices = _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi") + assert voices == [None] + + +def test_an_unnamed_voice_still_reports_something_resettable(monkeypatch, ctrl): + """voice_name is optional server-side — falling back to the id keeps the + tray entry from reading "now: " with nothing after it.""" + _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1")) + assert ctrl.current_voice() == "VOICE1" + + +def test_petctl_voice_reset_returns_bolt_to_his_own_voice(monkeypatch, ctrl): + """The server can pick a voice but can't ask for the default back — it + was never told what Bolt's own voice id is. This is how it asks.""" + ran = [] + monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd)) + _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence")) + + output = ctrl._handle_command("petctl voice reset") + + assert ran == [] # never reaches a shell, like every other petctl verb + assert "Terence" in output # the server can't see the voice; tell it what changed + assert ctrl.current_voice() == "" + assert _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi") == [None] + + +def test_petctl_voice_reset_says_so_when_there_was_nothing_to_reset(ctrl): + assert "already" in ctrl._handle_command("petctl voice reset") + + +# ── dialoguectl (multi-voice scenes) ──────────────────────────────────────── + +def _dialogue_command(*lines): + import json + return "dialoguectl " + json.dumps({"lines": list(lines)}) + + +def test_dialoguectl_never_reaches_the_shell(monkeypatch, ctrl): + ran = [] + monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd)) + monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC") + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", + lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000)) + monkeypatch.setattr(controller_mod.tts, "play_pcm", + lambda pcm, rate, should_stop=None: True) + + output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "[cheerfully] hi"})) + + assert ran == [] + assert "[dialogue] played 1 line" in output + + +def test_a_scene_shows_in_the_bubble_with_the_delivery_tags_stripped(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC") + monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x") + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", + lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000)) + monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True) + said = _capture(ctrl.said) + + ctrl._handle_command(_dialogue_command( + {"voice": "self", "text": "[cheerfully] Hello there!"}, + {"voice": "narrator", "text": "[whispering] He is lying."}, + )) + + assert said == ["Hello there! He is lying."] + assert ctrl.history.last().text == "Hello there! He is lying." + + +def test_a_mid_turn_scene_returns_to_thinking_not_idle(monkeypatch, ctrl): + """The server is still waiting on the tool result, so the pet talks and + goes back to waiting — dropping to IDLE would look like the turn ended.""" + monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC") + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", + lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000)) + monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True) + ctrl._state.transition(PetState.LISTENING) + ctrl._state.transition(PetState.THINKING) + states = _capture(ctrl.state_changed) + + ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"})) + + assert states == ["talking", "thinking"] + assert ctrl._state.state == PetState.THINKING + + +def test_the_scene_uses_a_voice_the_server_picked_with_speak_as(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC") + seen = {} + + def capture(inputs, model_id=None, stability=None): + seen["inputs"] = inputs + return np.zeros(4, dtype=np.int16), 24000 + + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", capture) + monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True) + ctrl._apply_voice(controller_mod.server_client.Reply("ok", "PICKEDvoice123456789", "Terence")) + + ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"})) + + assert seen["inputs"][0]["voice_id"] == "PICKEDvoice123456789" + + +def test_a_synthesis_failure_is_reported_back_for_bolt_to_retry(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC") + + def boom(inputs, model_id=None, stability=None): + raise controller_mod.tts.TtsError("voice_id not found") + + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", boom) + + output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"})) + + assert "couldn't synthesize" in output and "voice_id not found" in output + assert ctrl._state.state == PetState.IDLE # nothing left half-transitioned + + +def test_a_bad_voice_name_comes_back_as_advice_not_an_exception(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x") + output = ctrl._handle_command(_dialogue_command({"voice": "wizard", "text": "hi"})) + assert "unknown voice" in output and "narrator" in output + + +def test_dialogue_can_be_switched_off_on_this_device(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "DIALOGUE", False) + called = [] + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", + lambda *a, **k: called.append(1)) + output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"})) + assert "disabled" in output and called == [] + + +def test_talking_over_a_scene_is_reported_up_the_relay(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC") + monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", + lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000)) + monkeypatch.setattr(controller_mod.tts, "play_pcm", + lambda pcm, rate, should_stop=None: False) # barge-in + + output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"})) + + assert "interrupted" in output + + +# ── petctl self_restart ───────────────────────────────────────────────────── + +def test_self_restart_arms_after_the_turn_rather_than_dying_mid_relay(monkeypatch, ctrl, tmp_path): + """Restarting inline would kill the HTTP tool relay before the result was + posted, and the server would wait out its timeout on a turn that can never + finish. So the command returns, the turn completes, *then* the pet dies.""" + monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json") + monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None) + restarts = _capture(ctrl.restart_requested) + + output = ctrl._handle_command("petctl self_restart check the new dialogue code") + + assert "restarting as soon as this turn finishes" in output + assert restarts == [] # nothing has happened yet + + assert ctrl._maybe_self_restart() is True + assert restarts and "check the new dialogue code" in restarts[0] + + +def test_a_broken_edit_is_reported_instead_of_leaving_nothing_running(monkeypatch, ctrl, tmp_path): + monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json") + + def boom(*args, **kwargs): + raise controller_mod.self_restart.RestartError( + "the current code does not import, so restarting would leave you with " + "nothing running. Fix this first:\nSyntaxError: invalid syntax" + ) + + monkeypatch.setattr(controller_mod.self_restart, "preflight", boom) + restarts = _capture(ctrl.restart_requested) + + output = ctrl._handle_command("petctl self_restart try the new code") + + assert "SyntaxError" in output and "refused" in output + assert ctrl._maybe_self_restart() is False + assert restarts == [] + + +def test_a_second_restart_request_in_one_turn_is_a_no_op(monkeypatch, ctrl, tmp_path): + monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json") + monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None) + ctrl._handle_command("petctl self_restart first") + assert "already armed" in ctrl._handle_command("petctl self_restart second") + + +def test_self_restart_can_be_switched_off_on_this_device(monkeypatch, ctrl): + monkeypatch.setattr(controller_mod.config, "SELF_RESTART", False) + checked = [] + monkeypatch.setattr(controller_mod.self_restart, "preflight", + lambda *a, **k: checked.append(1)) + assert "disabled" in ctrl._handle_command("petctl self_restart go") + assert checked == [] + + +def test_coming_back_up_reports_to_the_server_and_speaks_the_reply(monkeypatch, ctrl, tmp_path): + """The half that makes it a loop: the new process tells Bolt it's back and + why, and his answer is spoken like any other turn.""" + state = tmp_path / "ctx.json" + monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", state) + controller_mod.self_restart.arm("check the walk cycle", version="0.2.3", + path=state, now=1000.0) + sent, spoken = [], [] + monkeypatch.setattr(controller_mod.server_client, "converse", + lambda text, on_command=None: sent.append(text) or Reply("Good, it's up.")) + monkeypatch.setattr(controller_mod.tts, "speak", + lambda text, on_error=None, should_stop=None, voice_id=None: + spoken.append(text) or True) + + ctrl._report_self_restart() + + assert "[pet self-restart]" in sent[0] and "check the walk cycle" in sent[0] + assert spoken == ["Good, it's up."] + # Consumed, so the next start doesn't announce the same restart again. + assert controller_mod.self_restart.load(state) is None + + +def test_an_ordinary_start_reports_nothing(monkeypatch, ctrl, tmp_path): + monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "none.json") + called = [] + monkeypatch.setattr(controller_mod.server_client, "converse", + lambda text, on_command=None: called.append(text)) + + ctrl._report_self_restart() + + assert called == [] diff --git a/tests/test_dialogue.py b/tests/test_dialogue.py new file mode 100644 index 0000000..8dadfd7 --- /dev/null +++ b/tests/test_dialogue.py @@ -0,0 +1,225 @@ +"""`dialoguectl` — multi-voice scene parsing, voice resolution, API limits, +and the request the ElevenLabs Text to Dialogue endpoint actually gets. + +Pure logic plus one mocked HTTP call: no audio device, no network, no display. +""" + +import sys +from pathlib import Path +from unittest.mock import MagicMock, patch + +import numpy as np +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from bolt_pet import dialogue +from bolt_pet.audio import tts + +SELF_ID = "aaorr6ZHIL88gEexu7dC" +NARRATOR_ID = "9BWtsMINqrJLrRacOk9x" +VILLAIN_ID = "IKne3meq5aSn9XLyUdCD" +VOICES = {"narrator": NARRATOR_ID, "villain": VILLAIN_ID} + + +def _scene(*lines): + return '{"lines": [' + ", ".join(lines) + "]}" + + +# ── parsing ───────────────────────────────────────────────────────────────── + +def test_non_dialogue_commands_are_left_alone(): + assert dialogue.parse("ls -la") is None + assert dialogue.parse('filectl {"op": "list"}') is None + assert dialogue.parse("") is None + # "dialogues" must not be mistaken for the "dialogue" prefix + assert dialogue.parse("dialogues --list") is None + + +def test_a_scene_parses_into_lines(): + action = dialogue.parse( + 'dialoguectl ' + _scene( + '{"voice": "self", "text": "[cheerfully] Hello, how are you?"}', + '{"voice": "villain", "text": "[stuttering] I am... fine."}', + ) + ) + assert action["action"] == "dialogue" + assert [line["voice"] for line in action["lines"]] == ["self", "villain"] + assert action["lines"][0]["text"].startswith("[cheerfully]") + + +def test_the_elevenlabs_field_names_are_accepted_too(): + """The model has read that API; copying its shape is the obvious thing to + try, so 'inputs'/'voice_id' work as well as 'lines'/'voice'.""" + action = dialogue.parse( + 'dialoguectl {"inputs": [{"voice_id": "%s", "text": "hi"}]}' % NARRATOR_ID + ) + assert action["lines"] == [{"voice": NARRATOR_ID, "text": "hi"}] + + +def test_a_line_with_no_voice_defaults_to_the_pet_itself(): + action = dialogue.parse('dialoguectl {"lines": [{"text": "just me talking"}]}') + assert action["lines"][0]["voice"] == "self" + + +def test_truncated_json_explains_the_one_line_rule(): + """The real failure mode: the server's command extractor stops at the + first newline, so a multi-line payload arrives cut in half. The error has + to name the cause, since Bolt is the one who has to fix it.""" + with pytest.raises(dialogue.DialogueError, match="one line"): + dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"') + + +def test_an_empty_or_shapeless_payload_is_rejected(): + with pytest.raises(dialogue.DialogueError, match="needs a JSON argument"): + dialogue.parse("dialoguectl") + with pytest.raises(dialogue.DialogueError, match="non-empty"): + dialogue.parse('dialoguectl {"lines": []}') + with pytest.raises(dialogue.DialogueError, match="no text"): + dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": " "}]}') + + +def test_optional_model_and_stability_ride_along(): + action = dialogue.parse( + 'dialoguectl {"model_id": "eleven_v3", "stability": 0.8, ' + '"lines": [{"text": "hi"}]}' + ) + assert action["model"] == "eleven_v3" + assert action["stability"] == 0.8 + + +# ── voice resolution ──────────────────────────────────────────────────────── + +def test_named_voices_resolve_from_the_configured_cast(): + action = dialogue.parse('dialoguectl ' + _scene( + '{"voice": "narrator", "text": "Once upon a time."}', + '{"voice": "villain", "text": "Not this again."}', + )) + inputs = dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID) + assert [entry["voice_id"] for entry in inputs] == [NARRATOR_ID, VILLAIN_ID] + + +def test_self_tracks_the_voice_the_pet_is_currently_using(): + """A scene featuring Bolt should sound like whoever Bolt currently is — + including a voice the server picked mid-conversation with speak_as.""" + action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}') + picked = "VOICEfromSPEAKas1234" + assert dialogue.resolve(action, self_voice=picked)[0]["voice_id"] == picked + + +def test_a_raw_voice_id_passes_straight_through(): + action = dialogue.parse('dialoguectl {"lines": [{"voice": "%s", "text": "hi"}]}' % NARRATOR_ID) + assert dialogue.resolve(action, self_voice=SELF_ID)[0]["voice_id"] == NARRATOR_ID + + +def test_an_unknown_name_lists_what_is_available(): + action = dialogue.parse('dialoguectl {"lines": [{"voice": "wizard", "text": "hi"}]}') + with pytest.raises(dialogue.DialogueError) as excinfo: + dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID) + message = str(excinfo.value) + assert "wizard" in message and "narrator" in message and "villain" in message + + +def test_self_without_a_configured_voice_says_so(): + action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}') + with pytest.raises(dialogue.DialogueError, match="ELEVENLABS_VOICE_ID"): + dialogue.resolve(action, self_voice="") + + +def test_the_voice_map_parser_skips_typos_instead_of_dying(): + voices = dialogue.parse_voice_map(f"narrator:{NARRATOR_ID}, broken-entry, villain:{VILLAIN_ID}") + assert voices == {"narrator": NARRATOR_ID, "villain": VILLAIN_ID} + assert dialogue.parse_voice_map("") == {} + + +# ── API limits, enforced before the request goes out ──────────────────────── + +def test_too_many_distinct_voices_is_refused_locally(): + inputs = [{"text": "hi", "voice_id": f"voice{index:015d}"} for index in range(11)] + with pytest.raises(dialogue.DialogueError, match="limit is 10"): + dialogue.check_limits(inputs) + + +def test_an_over_long_scene_is_refused_with_advice(): + inputs = [{"text": "x" * 1100, "voice_id": SELF_ID} for _ in range(2)] + with pytest.raises(dialogue.DialogueError) as excinfo: + dialogue.check_limits(inputs) + assert "Split it" in str(excinfo.value) # actionable, since Bolt reads this + + +# ── display / reporting ───────────────────────────────────────────────────── + +def test_delivery_tags_are_stripped_from_what_the_bubble_shows(): + action = dialogue.parse('dialoguectl ' + _scene( + '{"voice": "self", "text": "[cheerfully] Hello there!"}', + '{"voice": "narrator", "text": "[whispering] He is lying."}', + )) + assert dialogue.spoken_text(action) == "Hello there! He is lying." + + +def test_the_relay_report_names_the_cast(): + action = dialogue.parse('dialoguectl ' + _scene( + '{"voice": "self", "text": "one"}', '{"voice": "narrator", "text": "two"}', + )) + assert dialogue.describe(action) == "[dialogue] played 2 lines in 2 voices: narrator, self" + + +# ── the HTTP request ──────────────────────────────────────────────────────── + +def _pcm_response(samples=(1, 2, 3, 4)): + response = MagicMock() + response.content = np.array(samples, dtype=np.int16).tobytes() + response.raise_for_status = MagicMock() + return response + + +def test_the_request_matches_the_text_to_dialogue_api(monkeypatch): + monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key") + monkeypatch.setattr(tts.config, "TTS_SAMPLE_RATE", 24000) + monkeypatch.setattr(tts.config, "DIALOGUE_MODEL_ID", "eleven_v3") + inputs = [ + {"text": "[cheerfully] Hello", "voice_id": NARRATOR_ID}, + {"text": "[stuttering] H-hi", "voice_id": VILLAIN_ID}, + ] + with patch.object(tts.requests, "post", return_value=_pcm_response()) as post: + pcm, rate = tts.synthesize_dialogue(inputs) + + assert rate == 24000 and pcm.tolist() == [1, 2, 3, 4] + args, kwargs = post.call_args + assert args[0] == "https://api.elevenlabs.io/v1/text-to-dialogue" + assert kwargs["params"] == {"output_format": "pcm_24000"} + assert kwargs["headers"] == {"xi-api-key": "test-key"} + assert kwargs["json"]["inputs"] == inputs + assert kwargs["json"]["model_id"] == "eleven_v3" + assert "settings" not in kwargs["json"] # omitted unless asked for + + +def test_stability_is_only_sent_when_given(monkeypatch): + monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key") + with patch.object(tts.requests, "post", return_value=_pcm_response()) as post: + tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}], stability=0.3) + assert post.call_args.kwargs["json"]["settings"] == {"stability": 0.3} + + +def test_a_rejected_request_surfaces_what_the_api_said(monkeypatch): + """The API explains refusals in the body; Bolt reads this through the tool + relay, so it has to reach him rather than being flattened to '422'.""" + monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key") + failure = MagicMock() + failure.text = '{"detail": "voice_id not found"}' + error = Exception("422 Client Error") + error.response = failure + response = MagicMock() + response.raise_for_status = MagicMock(side_effect=error) + + with patch.object(tts.requests, "post", return_value=response): + with pytest.raises(tts.TtsError, match="voice_id not found"): + tts.synthesize_dialogue([{"text": "hi", "voice_id": "nope"}]) + + +def test_no_api_key_fails_before_the_request(monkeypatch): + monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "") + with patch.object(tts.requests, "post") as post: + with pytest.raises(tts.TtsError): + tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}]) + post.assert_not_called() diff --git a/tests/test_pet_actions.py b/tests/test_pet_actions.py index 9d90e2b..c321e9c 100644 --- a/tests/test_pet_actions.py +++ b/tests/test_pet_actions.py @@ -69,3 +69,17 @@ def test_describe_is_reported_back_to_the_server(): assert "top-left" in pet_actions.describe({"action": "move", "anchor": "top-left"}) assert "wave" in pet_actions.describe({"action": "emote", "emote": "wave"}) assert pet_actions.describe({"action": "help"}) == pet_actions.HELP + + +def test_voice_reset_parses_with_or_without_the_word_reset(): + assert pet_actions.parse("petctl voice reset") == {"action": "voice", "voice": "default"} + assert pet_actions.parse("petctl voice default") == {"action": "voice", "voice": "default"} + assert pet_actions.parse("petctl voice") == {"action": "voice", "voice": "default"} + + +def test_petctl_cannot_be_used_to_pick_a_voice(): + """Choosing a voice is the server's job (speak_as) — it has the voice + library. petctl only ever undoes one, so an attempt to set a voice here + is pointed back at the marker that works.""" + with pytest.raises(pet_actions.ActionError, match="speak_as"): + pet_actions.parse("petctl voice Terence") diff --git a/tests/test_self_restart.py b/tests/test_self_restart.py new file mode 100644 index 0000000..1f9de64 --- /dev/null +++ b/tests/test_self_restart.py @@ -0,0 +1,158 @@ +"""`petctl self_restart` — the pet restarting itself and remembering why. + +Everything here runs against a temp context file and a fake subprocess runner, +so the tests exercise the arming/preflight/report logic without any process +actually dying. +""" + +import subprocess +import sys +from pathlib import Path +from types import SimpleNamespace + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from bolt_pet import pet_actions, self_restart + + +@pytest.fixture +def state(tmp_path): + return tmp_path / "restart_context.json" + + +def _ok_run(*args, **kwargs): + return SimpleNamespace(returncode=0, stdout="", stderr="") + + +def _broken_run(*args, **kwargs): + return SimpleNamespace( + returncode=1, stdout="", + stderr=' File "bolt_pet/controller.py", line 42\n def _speak(\nSyntaxError: invalid syntax', + ) + + +# ── parsing ───────────────────────────────────────────────────────────────── + +def test_self_restart_parses_with_a_free_text_reason(): + action = pet_actions.parse("petctl self_restart check the new walk cycle loads") + assert action == { + "action": "self_restart", "reason": "check the new walk cycle loads", + } + + +def test_self_restart_needs_no_reason_and_accepts_aliases(): + assert pet_actions.parse("petctl self_restart")["reason"] == "" + assert pet_actions.parse("petctl restart")["action"] == "self_restart" + assert pet_actions.parse("petctl reboot")["action"] == "self_restart" + + +def test_self_restart_is_listed_in_the_help(): + assert "self_restart" in pet_actions.HELP + + +# ── preflight ─────────────────────────────────────────────────────────────── + +def test_preflight_passes_when_the_code_imports(): + self_restart.preflight(run=_ok_run) # no exception + + +def test_preflight_hands_back_the_traceback_instead_of_dying(state): + """The whole point: a syntax error Bolt just introduced comes back as + something he can read and fix, in the same turn, with the pet still up.""" + with pytest.raises(self_restart.RestartError) as excinfo: + self_restart.preflight(run=_broken_run) + message = str(excinfo.value) + assert "does not import" in message + assert "SyntaxError" in message and "controller.py" in message + + +def test_preflight_runs_the_import_in_a_subprocess_not_here(): + """This process holds the *old* modules, so an in-process import would + pass on a file that no longer parses.""" + seen = {} + + def capture(cmd, **kwargs): + seen["cmd"], seen["kwargs"] = cmd, kwargs + return SimpleNamespace(returncode=0, stdout="", stderr="") + + self_restart.preflight(run=capture) + assert seen["cmd"][0] == sys.executable + assert "import bolt_pet" in seen["cmd"][2] + assert seen["kwargs"]["env"]["QT_QPA_PLATFORM"] == "offscreen" # imports need no display + + +def test_a_subprocess_that_cannot_even_run_is_reported(monkeypatch): + def explode(*args, **kwargs): + raise OSError("no python here") + + with pytest.raises(self_restart.RestartError, match="couldn't run the preflight"): + self_restart.preflight(run=explode) + + +# ── context across the restart ────────────────────────────────────────────── + +def test_arming_persists_the_reason_for_the_next_process(state): + self_restart.arm("check the sprite frames load", version="0.2.3", + session="pet-desktop", recent=["you: reload the sprites"], + path=state, now=1000.0) + revived = self_restart.load(state) + assert revived.reason == "check the sprite frames load" + assert revived.version == "0.2.3" + assert revived.recent == ["you: reload the sprites"] + assert revived.restarts == [1000.0] + + +def test_no_context_means_a_normal_start(state): + assert self_restart.load(state) is None + + +def test_a_corrupt_context_file_is_ignored_not_fatal(state): + state.write_text("{not json at all", encoding="utf-8") + assert self_restart.load(state) is None + + +def test_clearing_the_context_stops_it_being_re_announced(state): + self_restart.arm("once", path=state, now=1000.0) + self_restart.clear(state) + assert self_restart.load(state) is None + self_restart.clear(state) # clearing twice is not an error + + +def test_the_report_says_what_happened_and_what_to_check(state): + context = self_restart.arm( + "verify the dialogue command works", verify="verify the dialogue command works", + version="0.2.3", recent=["you: try a scene"], path=state, now=1000.0, + ) + text = self_restart.report(context, version="0.2.4", now=1004.5) + assert "I restarted myself" in text + assert "verify the dialogue command works" in text + assert "4.5s" in text + assert "0.2.4" in text and "was 0.2.3" in text + assert "you: try a scene" in text + + +# ── loop guard ────────────────────────────────────────────────────────────── + +def test_restart_history_accumulates_across_restarts(state): + self_restart.arm("one", path=state, now=1000.0) + self_restart.arm("two", path=state, now=1100.0) + assert self_restart.load(state).restarts == [1000.0, 1100.0] + + +def test_too_many_restarts_in_the_window_is_refused(state): + now = 1000.0 + for index in range(self_restart.MAX_RESTARTS): + self_restart.arm(f"attempt {index}", path=state, now=now + index) + with pytest.raises(self_restart.RestartError, match="looping"): + self_restart.check_loop_guard(self_restart.load(state), now=now + 10) + + +def test_old_restarts_fall_out_of_the_window(state): + now = 1000.0 + for index in range(self_restart.MAX_RESTARTS): + self_restart.arm(f"attempt {index}", path=state, now=now + index) + later = now + self_restart.WINDOW_SECONDS + 60 + self_restart.check_loop_guard(self_restart.load(state), now=later) # no exception + assert self_restart.recent_restarts(self_restart.load(state), now=later) == [] diff --git a/tests/test_server_client.py b/tests/test_server_client.py index 6f7c87c..7c3b8a7 100644 --- a/tests/test_server_client.py +++ b/tests/test_server_client.py @@ -27,7 +27,8 @@ def test_converse_returns_reply_directly(): with patch.object(server_client.requests, "post") as post: post.return_value = _mock_response({"type": "reply", "text": "hello there"}) result = server_client.converse("hi") - assert result == "hello there" + assert result.text == "hello there" + assert result.voice_id == "" # no speak_as on this reply post.assert_called_once() args, kwargs = post.call_args assert args[0] == "http://test-server:5002/desk/converse" @@ -43,7 +44,7 @@ def test_converse_relays_a_command_then_returns_reply(): with patch.object(server_client.requests, "post", side_effect=responses) as post: on_command = MagicMock(return_value="[exit 0]\nhi") result = server_client.converse("run echo hi", on_command=on_command) - assert result == "done" + assert result.text == "done" on_command.assert_called_once_with("echo hi") # second call was to /desk/tool_result with the command's output second_call = post.call_args_list[1] @@ -53,6 +54,19 @@ def test_converse_relays_a_command_then_returns_reply(): } +def test_converse_carries_a_speak_as_voice_back_with_the_reply(): + """The server tags a reply with the voice it picked (speak_as); this + client is what actually speaks in it, so the id has to survive the + return trip rather than being dropped with the rest of the payload.""" + with patch.object(server_client.requests, "post") as post: + post.return_value = _mock_response({ + "type": "reply", "text": "Ahoy there.", + "voice_id": "hnhGxwvHP8fc469w51rM", "voice_name": "Terence", + }) + result = server_client.converse("talk like a pirate") + assert result == server_client.Reply("Ahoy there.", "hnhGxwvHP8fc469w51rM", "Terence") + + def test_converse_raises_server_error_on_error_payload(): with patch.object(server_client.requests, "post") as post: post.return_value = _mock_response({"type": "error", "error": "unauthorized"}) diff --git a/tests/test_tts_stream.py b/tests/test_tts_stream.py index af78d04..c5608fc 100644 --- a/tests/test_tts_stream.py +++ b/tests/test_tts_stream.py @@ -8,7 +8,8 @@ import numpy as np sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) -from bolt_pet.audio.tts import chunks_to_int16 +from bolt_pet import config as tts_config +from bolt_pet.audio.tts import chunks_to_int16, model_for, voice_for from bolt_pet.audio.wake_word import NearMissLog @@ -80,3 +81,27 @@ def test_clear_resets_peak_and_entries(): log.observe(0.45, threshold=0.5, timestamp=1.0) log.clear() assert log.entries() == [] and log.peak == 0.0 + + +# ── voice / model selection (server speak_as) ─────────────────────────────── + +def test_the_override_voice_wins_over_the_configured_one(monkeypatch): + monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "DEFAULT") + assert voice_for("VOICE1") == "VOICE1" + assert voice_for("") == "DEFAULT" + assert voice_for(None) == "DEFAULT" + + +def test_english_replies_in_the_default_voice_use_the_default_model(monkeypatch): + monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2") + monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5") + assert model_for("all good here", None) == "eleven_flash_v2" + + +def test_a_picked_voice_or_non_english_text_uses_the_multilingual_model(monkeypatch): + # eleven_flash_v2 is English-only: it would read either of these as + # mangled phonetic English rather than failing outright. + monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2") + monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5") + assert model_for("all good here", "VOICE1") == "eleven_flash_v2_5" + assert model_for("こんにちは", None) == "eleven_flash_v2_5"