Add text-to-dialogue, self-restart capability, and misc updates
This commit is contained in:
@@ -52,7 +52,38 @@
|
||||
"Bash(docker exec bolt *)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen /root/Documents/bolt-pet/.venv/bin/pytest /home/themajesticmagician/Documents/Bolt-Pet/tests/test_file_ops.py -q)",
|
||||
"Bash(grep -rn *)",
|
||||
"Bash(git ls-tree *)"
|
||||
"Bash(git ls-tree *)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_controller_features.py -q)",
|
||||
"Read(//home/maji/Documents/tmn-api/**)",
|
||||
"Bash(timeout 300 .venv/bin/python -m pytest tests/test_desk_voice.py -q)",
|
||||
"Bash(echo \"exit=$?\")",
|
||||
"Bash(timeout 300 /home/maji/Documents/tmn-api/.venv/bin/python -m pytest /home/maji/Documents/tmn-api/tests/test_desk_voice.py -q -p no:cacheprovider --rootdir=/home/maji/Documents/tmn-api)",
|
||||
"Bash(echo \"EXIT=$?\")",
|
||||
"Bash(/home/maji/Documents/tmn-api/.venv/bin/python -c \"import ast,pathlib; ast.parse\\(pathlib.Path\\('/home/maji/Documents/tmn-api/ai/desk_api.py'\\).read_text\\(\\)\\); print\\('desk_api.py parses OK'\\)\")",
|
||||
"Bash(ps -eo pid,etime,cmd)",
|
||||
"Bash(systemctl --user list-units --type=service)",
|
||||
"Read(//run/user/1000/gvfs/sftp:host=192.168.2.231,user=root/Main/Docker-Compose/TMN-API/tmn-api/ai/**)",
|
||||
"Bash(findmnt -T /home/maji/Documents/tmn-api -o TARGET,SOURCE,FSTYPE)",
|
||||
"Bash(echo \"rc=$?\")",
|
||||
"Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py -q -p no:cacheprovider)",
|
||||
"Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py -q -p no:cacheprovider)",
|
||||
"Bash(awk 'NR>=1150 && NR<=1310 && \\(/return / || /def /\\)' ai/agents/default.py)",
|
||||
"Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python -c ' *)",
|
||||
"Bash(timeout 900 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py -q -p no:cacheprovider)",
|
||||
"Bash(timeout 300 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_proactive.py::test_send_trims_swallowed_tool_lines -q -p no:cacheprovider)",
|
||||
"Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python *)",
|
||||
"Bash(./tmnvenv/bin/pip install *)",
|
||||
"Bash(./tmnvenv/bin/python -c \"import pytest,dotenv,yaml; print\\('scratch venv ready'\\)\")",
|
||||
"Bash(/tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/pip install *)",
|
||||
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py)",
|
||||
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider)",
|
||||
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py)",
|
||||
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_default_agent.py -q -p no:cacheprovider)",
|
||||
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py -q -p no:cacheprovider)",
|
||||
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py tests/test_desk_api.py tests/test_main_helpers.py -q -p no:cacheprovider)",
|
||||
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py --ignore=tests/test_billing_web.py)",
|
||||
"WebFetch(domain:elevenlabs.io)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_dialogue.py -q)"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,6 +31,36 @@ ELEVENLABS_VOICE_ID=
|
||||
#ELEVENLABS_MODEL_ID=eleven_flash_v2
|
||||
#TTS_SAMPLE_RATE=24000
|
||||
|
||||
# Ask Bolt to use a different voice (or another language) and the server
|
||||
# picks one from the ElevenLabs voice library and tags the reply with it.
|
||||
# It only tags one reply, and it can't remember the id afterwards — so the
|
||||
# pet keeps using that voice until a new one is picked or you choose "Use
|
||||
# default voice" in the tray. VOICE_STICKY=false makes each pick last for
|
||||
# exactly the one reply it came with instead.
|
||||
#VOICE_STICKY=true
|
||||
|
||||
# ── Multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
|
||||
# Lets Bolt play a short scene in several voices with delivery tags the v3
|
||||
# model acts on ("[cheerfully] Hello", "[whispering] He is lying"), driven by
|
||||
# the server through a relayed `dialoguectl` command. Name the cast here —
|
||||
# "self" always means whatever voice the pet is currently using.
|
||||
#DIALOGUE=true
|
||||
#DIALOGUE_MODEL_ID=eleven_v3
|
||||
#DIALOGUE_VOICES=narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD
|
||||
|
||||
# ── Self-restart (optional) ─────────────────────────────────────────────────
|
||||
# `petctl self_restart <why>` lets Bolt reload the pet after editing its own
|
||||
# code, so he can check the change live. The code is import-checked first, the
|
||||
# restart waits for the current turn to finish, and the reason is carried
|
||||
# across so the new process reports back. The guard refuses more than
|
||||
# SELF_RESTART_MAX restarts within SELF_RESTART_WINDOW_SECONDS.
|
||||
#SELF_RESTART=true
|
||||
#SELF_RESTART_MAX=5
|
||||
#SELF_RESTART_WINDOW_SECONDS=900
|
||||
# Used instead of ELEVENLABS_MODEL_ID whenever the server picked the voice
|
||||
# or the reply has non-ASCII in it — the flash_v2 default is English-only.
|
||||
#ELEVENLABS_MULTILINGUAL_MODEL_ID=eleven_flash_v2_5
|
||||
|
||||
# ── Audio devices (optional — leave blank for the system default) ──────────
|
||||
#MIC_DEVICE=
|
||||
#SPEAKER_DEVICE=
|
||||
|
||||
@@ -14,7 +14,8 @@ Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk /
|
||||
click → record utterance → Deepgram STT → + active-window + screen-layout
|
||||
context → POST /desk/converse → [server may relay a shell command to run on
|
||||
this machine, or a `petctl` pseudo-command that moves/emotes the pet, jumps it
|
||||
to another monitor, or reads a screen's text back instead] → reply →
|
||||
to another monitor, reads a screen's text back, or plays a multi-voice scene
|
||||
instead] → reply (optionally tagged with a voice the server picked for it) →
|
||||
ElevenLabs streaming TTS (or offline pyttsx3 fallback) → speakers`, with the
|
||||
pet sprite/speech bubble reflecting state throughout, and playback
|
||||
interruptible by talking over it (barge-in).
|
||||
@@ -79,7 +80,10 @@ logs a missing-config message and exits its thread instead of starting.
|
||||
`/desk/tool_result` until the server sends a final `reply` (capped at
|
||||
`_MAX_RELAY_HOPS`). This is the same "full desktop control" trust model as
|
||||
the server repo's other desk clients — commands only ever originate from
|
||||
the user's own voice/click requests in their own session. `list_outbox_files`
|
||||
the user's own voice/click requests in their own session. A final reply is
|
||||
returned as a `Reply(text, voice_id, voice_name)` rather than a bare string,
|
||||
because the server can tag it with a voice — see "Voices" below.
|
||||
`list_outbox_files`
|
||||
/ `download_outbox_file` hit the same `/desk/files` and `/desk/files/<id>`
|
||||
endpoints the server's `deliver_files` tool queues onto — see `file_delivery.py`.
|
||||
- **`file_delivery.py`** — the filesystem half of receiving files the server
|
||||
@@ -104,7 +108,11 @@ logs a missing-config message and exits its thread instead of starting.
|
||||
`play_stream()` start playback on the first chunk; `chunks_to_int16()`
|
||||
carries odd bytes across HTTP chunk boundaries, without which everything
|
||||
after the first split sample plays as static — falling back to whole-clip
|
||||
PCM then offline `pyttsx3`), `barge_in.py` (two detectors behind one
|
||||
PCM then offline `pyttsx3`; every entry point takes an optional `voice_id`
|
||||
overriding `ELEVENLABS_VOICE_ID`, and `model_for()` picks the multilingual
|
||||
model whenever there's an override or non-ASCII text, since the default
|
||||
`eleven_flash_v2` is English-only and would read either as garbled
|
||||
phonetic English rather than failing), `barge_in.py` (two detectors behind one
|
||||
`reset()`/`check()` shape, chosen by `BARGE_IN_MODE` via `make_detector`:
|
||||
**wake** (default) scores every frame with the same openWakeWord model the
|
||||
idle listener uses, so only the wake phrase cuts playback; **energy** is the
|
||||
@@ -115,8 +123,8 @@ logs a missing-config message and exits its thread instead of starting.
|
||||
accepts an injectable stream/model/protocol so tests don't need real audio
|
||||
hardware or a display.
|
||||
- **`pet_actions.py`** — `petctl` pseudo-commands (`petctl move top-left`,
|
||||
`petctl emote wave`, `say`/`wander`/`nap`, plus the screen verbs
|
||||
`jump`/`monitors`/`read`). The desk API has no "move the pet" payload type
|
||||
`petctl emote wave`, `say`/`wander`/`nap`, the screen verbs
|
||||
`jump`/`monitors`/`read`, and `voice reset`). The desk API has no "move the pet" payload type
|
||||
and this repo can't change the server, so these ride the existing
|
||||
shell-command relay: `controller._handle_command` parses them and they never
|
||||
reach `subprocess`; anything else is a real shell command exactly as before.
|
||||
@@ -125,7 +133,51 @@ logs a missing-config message and exits its thread instead of starting.
|
||||
module doesn't have, so the spec passes through to `monitors.resolve()`.
|
||||
Query verbs (`monitors`, `read`) are answered in `_handle_command` rather
|
||||
than by `pet_actions.describe()`, because their output *is* the point: it
|
||||
goes back up the tool-result relay for Bolt to use in his reply.
|
||||
goes back up the tool-result relay for Bolt to use in his reply — as is
|
||||
`voice reset`, which reports what it dropped since the server can't see
|
||||
which voice is in use. `voice` only ever resets: picking one is the
|
||||
server's job (`speak_as`, which it already knows how to use), so a
|
||||
`petctl voice <name>` attempt is an error pointing back at that marker.
|
||||
- **`self_restart.py`** — `petctl self_restart`, the pet restarting itself so
|
||||
Bolt can *see* a code change he just made instead of waiting for a human to
|
||||
restart it. Three problems shape it, and all three are the interesting part.
|
||||
(1) The restart can't happen inline: killing the process mid-turn would drop
|
||||
the HTTP tool relay before the result was posted, leaving the server to wait
|
||||
out its timeout on a turn that can never finish — so the command only
|
||||
*arms* it (`controller._arm_self_restart`) and
|
||||
`controller._maybe_self_restart` fires it after the reply is spoken, the
|
||||
same "only between turns" rule the updater follows. (2) A broken edit must
|
||||
not be fatal, so `preflight()` imports the package in a **subprocess**
|
||||
before arming — this process holds the old modules, so an in-process import
|
||||
would pass on a file that no longer parses — and a SyntaxError comes back as
|
||||
the command's output, in the same turn, with the pet still running. (3) The
|
||||
reason has to outlive the process, so it's written to
|
||||
`~/.cache/bolt-pet/restart_context.json` (never inside the repo Bolt is
|
||||
editing) and read on the way back up by `controller._report_self_restart`,
|
||||
which posts it to the server as an ordinary turn — that's what makes
|
||||
"restart and check the sprites load" finish as a spoken sentence rather than
|
||||
a silence. `check_loop_guard` refuses after `SELF_RESTART_MAX` restarts in
|
||||
`SELF_RESTART_WINDOW_SECONDS`, so an edit-restart-crash cycle stops itself.
|
||||
Off switch: `SELF_RESTART=false`.
|
||||
- **`dialogue.py`** — `dialoguectl` pseudo-commands: a multi-voice *scene*
|
||||
through ElevenLabs' Text to Dialogue endpoint (`audio/tts.
|
||||
synthesize_dialogue`), checked in `_handle_command` between petctl and
|
||||
filectl. Same single-line-JSON wire format as filectl and for the same
|
||||
reason (the server's `command` marker captures only up to the next
|
||||
newline), and it accepts the ElevenLabs field names (`inputs`/`voice_id`)
|
||||
as well as its own (`lines`/`voice`) because the model has read that API
|
||||
and copying its shape is the obvious thing to try. Voices are *named*
|
||||
(`DIALOGUE_VOICES` maps names to ids) rather than pasted as raw ids, and
|
||||
`self` resolves to whatever voice the pet is speaking with right now —
|
||||
including a `speak_as` pick — so Bolt sounds like himself in his own
|
||||
scenes. The API's limits (10 distinct voices, ~2000 characters) are
|
||||
enforced *before* the request so a mistake comes back up the tool-result
|
||||
relay as a sentence Bolt can act on rather than an HTTP 422 he can't see.
|
||||
Unlike the normal reply path there is no streaming variant, so a scene is
|
||||
whole-clip: `controller._play_dialogue` plays it with the same bubble,
|
||||
transcript and barge-in handling a spoken reply gets, and returns to
|
||||
THINKING afterwards (not IDLE) because the server is still waiting on the
|
||||
tool result — that leg is why `state.py` allows TALKING -> THINKING.
|
||||
- **`file_ops.py`** — `filectl` pseudo-commands, checked in `_handle_command`
|
||||
right after petctl and before falling through to a real shell command.
|
||||
Executing arbitrary commands already worked via the shell relay
|
||||
@@ -274,9 +326,38 @@ logs a missing-config message and exits its thread instead of starting.
|
||||
name rather than by `PetState`, with `has()` reporting whether a key is
|
||||
backed by real art so callers can decline a placeholder instead of trotting
|
||||
a blob across the desktop; `tray.py` is the system tray menu (talk now / mute / nap / wander /
|
||||
click-through / history / wake-word tuning / quit) — the pet window has no
|
||||
title bar or taskbar entry; `history_window.py` and `wake_tuner.py` are the
|
||||
two dialogs it opens.
|
||||
click-through / history / wake-word tuning / use-default-voice / quit) — the
|
||||
pet window has no title bar or taskbar entry; `history_window.py` and
|
||||
`wake_tuner.py` are the two dialogs it opens.
|
||||
|
||||
### Voices (the server's `speak_as`)
|
||||
|
||||
Ask Bolt to talk like someone else, or in another language, and the *server*
|
||||
does the picking: its desk-only `voice_search` marker browses the ElevenLabs
|
||||
voice library, and `speak_as: <voice_id>` on the final reply tags that reply
|
||||
with the chosen voice (adding a Voice Library pick to the ElevenLabs account
|
||||
first, so the id is usable by the time it reaches us). Nothing about that is
|
||||
this repo's to decide — all the client owes it is actually speaking in the
|
||||
voice it was handed: `converse()` returns it on `Reply`, `_apply_voice()`
|
||||
records it, and `_speak()` passes it to `tts.speak(voice_id=...)`.
|
||||
|
||||
Two things are decided *here*, though, because the server can't:
|
||||
|
||||
- **The voice sticks** (`VOICE_STICKY`, default on). The server tags one
|
||||
reply and strips the marker before storing the turn, so it never sees the
|
||||
id again — "keep talking like that" would send it searching for a voice all
|
||||
over again, and it'd likely land on a different one. Holding the id
|
||||
client-side is what makes the rest of the conversation stay in that voice.
|
||||
An untagged reply therefore never *changes* the voice; only a new
|
||||
`speak_as`, `VOICE_STICKY=false`, or a reset does.
|
||||
- **There's a way back.** Since the server was never told Bolt's own voice
|
||||
id, it can't ask for it back with `speak_as` — so reverting is local: the
|
||||
tray's **Use default voice** entry (enabled only while a picked voice is
|
||||
in use, kept in sync by the `voice_changed` signal), a restart, or
|
||||
`petctl voice reset`, which is what lets Bolt honour "go back to your
|
||||
normal voice" out loud. That last one needs the server's pet prompt block
|
||||
(`ai/desk_api.py`, `pet_tools`) to mention the verb, or the model never
|
||||
emits it — the desk API's prompt is where petctl is advertised.
|
||||
|
||||
### Wake-word detection
|
||||
|
||||
|
||||
@@ -69,8 +69,8 @@ limitations).
|
||||
speaks, and barging in starts your next turn immediately (`BARGE_IN`).
|
||||
- Right-click the tray icon for **Talk now**, **Mute mic**, **Nap**,
|
||||
**Wander around**, **Click through the pet**, **History…**, **Wake word
|
||||
tuning…** and **Quit** — the pet window itself has no title bar or taskbar
|
||||
entry.
|
||||
tuning…**, **Use default voice** and **Quit** — the pet window itself has
|
||||
no title bar or taskbar entry.
|
||||
- **Click the speech bubble** to copy what it just said; the tray's
|
||||
**History…** window keeps the last `HISTORY_LIMIT` turns.
|
||||
|
||||
@@ -80,8 +80,8 @@ limitations).
|
||||
listening/thinking/talking or while a bubble is up.
|
||||
- **Moves and emotes on command.** Bolt can relay `petctl move top-left`,
|
||||
`petctl emote wave|hop|spin|nod|shake`, `petctl say ...`, `petctl wander
|
||||
on|off`, `petctl nap on|off`. These are intercepted here and never reach a
|
||||
shell.
|
||||
on|off`, `petctl nap on|off`, `petctl voice reset`, and `dialoguectl` for a
|
||||
multi-voice scene. These are intercepted here and never reach a shell.
|
||||
- **Naps** during `QUIET_HOURS` (e.g. `23:00-08:00`) or while a fullscreen
|
||||
app is focused (`DND_ON_FULLSCREEN`) — it dims, stops wandering, and makes
|
||||
no proactive noise. It still answers when you speak to it.
|
||||
@@ -90,6 +90,41 @@ limitations).
|
||||
notifications get forwarded to the server, so it can tell you the deploy
|
||||
went green. Off by default: each one costs a round trip.
|
||||
|
||||
## Speaking in another voice
|
||||
|
||||
Ask for a different voice — "use a clearer voice", "talk like a pirate", "say
|
||||
that in Japanese" — and Bolt searches the ElevenLabs voice library on the
|
||||
server, picks one, and tags his reply with it (`speak_as`); the pet is what
|
||||
actually speaks in it. A Voice Library pick is added to your ElevenLabs
|
||||
account automatically the first time it's used, and non-English replies (or
|
||||
any picked voice) go through `ELEVENLABS_MULTILINGUAL_MODEL_ID` rather than
|
||||
the English-only `eleven_flash_v2` default.
|
||||
|
||||
The new voice **stays on** for the rest of the conversation, because the
|
||||
server tags a single reply and doesn't remember which voice it chose — so
|
||||
"keep talking like that" would otherwise send it hunting for a voice again.
|
||||
To get his own voice back: ask him ("use your normal voice" — he relays
|
||||
`petctl voice reset`), use **Use default voice** in the tray menu (greyed
|
||||
out unless a picked voice is active), or restart the pet. Set
|
||||
`VOICE_STICKY=false` in `.env` if you'd rather each pick lasted exactly one
|
||||
reply.
|
||||
|
||||
## Multi-voice dialogue
|
||||
|
||||
Ask for a scene — "do the argument between the two of them", "read that back
|
||||
as a radio play" — and Bolt can relay a `dialoguectl` command that the pet
|
||||
renders through ElevenLabs' Text to Dialogue endpoint: several voices in one
|
||||
take, with delivery tags the v3 model acts on (`[cheerfully]`, `[whispering]`,
|
||||
`[stuttering]`). One request per scene, so the voices actually react to each
|
||||
other instead of sounding like clips glued together.
|
||||
|
||||
Name the cast in `.env` (`DIALOGUE_VOICES=narrator:9BWts…,villain:IKne3…`);
|
||||
the name `self` always means whatever voice the pet is currently using, so
|
||||
Bolt sounds like himself in his own scenes — including after a `speak_as`
|
||||
switch. Scenes show up in the speech bubble with the tags stripped, count as
|
||||
normal speech for the transcript, and can be talked over like any other reply.
|
||||
`DIALOGUE=false` turns the whole thing off on this device.
|
||||
|
||||
## Wake-word detection
|
||||
|
||||
`bolt_pet/audio/wake_word.py` feeds every mic frame into `thunderbolt.onnx`
|
||||
|
||||
+96
-12
@@ -5,11 +5,16 @@ desk_client/bolt_desk.py which shells out because it only targets Linux.
|
||||
Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech
|
||||
on macOS, espeak on Linux) if ElevenLabs isn't configured or the request
|
||||
fails, so the pet can still talk with zero cloud config.
|
||||
|
||||
Every entry point takes an optional *voice_id* that overrides
|
||||
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
|
||||
reply marker reaches the speakers (see controller._apply_voice). The offline
|
||||
fallback has no such concept and always sounds like itself.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterable, Iterator
|
||||
from typing import Iterable, Iterator, Optional
|
||||
|
||||
import numpy as np
|
||||
import requests
|
||||
@@ -21,18 +26,40 @@ class TtsError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
|
||||
def voice_for(voice_id: Optional[str] = None) -> str:
|
||||
"""The voice this call should use: an override (server `speak_as`) if
|
||||
given, else the configured default."""
|
||||
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
|
||||
|
||||
|
||||
def model_for(text: str, voice_id: Optional[str] = None) -> str:
|
||||
"""Which ElevenLabs model to synthesize with.
|
||||
|
||||
The default (`eleven_flash_v2`) is English-only, and both things that
|
||||
reach this branch mean the reply probably isn't English: a voice the
|
||||
server picked mid-conversation is nearly always about a language or an
|
||||
accent, and non-ASCII text can't be English at all. Rendering either one
|
||||
through the English model gets you a mangled phonetic reading rather
|
||||
than a failure, which is worse — so those go through the multilingual
|
||||
model instead."""
|
||||
if (voice_id or "").strip() or not text.isascii():
|
||||
return config.ELEVENLABS_MULTILINGUAL_MODEL_ID
|
||||
return config.ELEVENLABS_MODEL_ID
|
||||
|
||||
|
||||
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
|
||||
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
|
||||
callers should fall back to speak_offline() rather than treating this
|
||||
as fatal."""
|
||||
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
|
||||
voice = voice_for(voice_id)
|
||||
if not (config.ELEVENLABS_API_KEY and voice):
|
||||
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
||||
try:
|
||||
response = requests.post(
|
||||
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}",
|
||||
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}",
|
||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
||||
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
|
||||
json={"text": text, "model_id": model_for(text, voice_id)},
|
||||
timeout=60,
|
||||
)
|
||||
response.raise_for_status()
|
||||
@@ -44,20 +71,23 @@ def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
|
||||
return pcm, config.TTS_SAMPLE_RATE
|
||||
|
||||
|
||||
def stream_pcm(text: str, chunk_bytes: int = 4096) -> Iterator[np.ndarray]:
|
||||
def stream_pcm(
|
||||
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
|
||||
) -> Iterator[np.ndarray]:
|
||||
"""Same audio as synthesize_pcm(), but yielded as it arrives from
|
||||
ElevenLabs' /stream endpoint so playback can start on the first chunk
|
||||
(~300ms) instead of after the whole clip is synthesized. Raises TtsError
|
||||
before yielding anything if the request itself fails, so callers can fall
|
||||
back cleanly; a mid-stream failure just ends the generator."""
|
||||
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
|
||||
voice = voice_for(voice_id)
|
||||
if not (config.ELEVENLABS_API_KEY and voice):
|
||||
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
||||
try:
|
||||
response = requests.post(
|
||||
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}/stream",
|
||||
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}/stream",
|
||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
||||
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
|
||||
json={"text": text, "model_id": model_for(text, voice_id)},
|
||||
timeout=60,
|
||||
stream=True,
|
||||
)
|
||||
@@ -83,6 +113,55 @@ def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
|
||||
yield np.frombuffer(data[:usable], dtype=np.int16)
|
||||
|
||||
|
||||
def synthesize_dialogue(
|
||||
inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None
|
||||
) -> tuple[np.ndarray, int]:
|
||||
"""Multi-voice scene via ElevenLabs Text to Dialogue.
|
||||
|
||||
One request, one take: the whole exchange is synthesized together, which
|
||||
is the point — the model hears the previous line, so reactions and timing
|
||||
land instead of sounding like separately-rendered clips.
|
||||
|
||||
Same PCM-over-`requests` posture as the rest of this module (no SDK, no
|
||||
`play()` shelling out to ffplay), so playback is the same sounddevice path
|
||||
everything else uses and barge-in works on it unchanged. There is no
|
||||
documented streaming variant, and a scene is a short set piece anyway, so
|
||||
this is whole-clip only.
|
||||
"""
|
||||
if not (config.ELEVENLABS_API_KEY and inputs):
|
||||
raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)")
|
||||
body: dict = {
|
||||
"inputs": [
|
||||
{"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")}
|
||||
for entry in inputs
|
||||
],
|
||||
"model_id": model_id or config.DIALOGUE_MODEL_ID,
|
||||
}
|
||||
if stability is not None:
|
||||
body["settings"] = {"stability": float(stability)}
|
||||
try:
|
||||
response = requests.post(
|
||||
"https://api.elevenlabs.io/v1/text-to-dialogue",
|
||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
||||
json=body,
|
||||
timeout=120, # a multi-voice take is slower to render than one line
|
||||
)
|
||||
response.raise_for_status()
|
||||
except Exception as exc:
|
||||
detail = ""
|
||||
# The API explains refusals (character limit, unknown voice) in the
|
||||
# body; surfacing it is what lets Bolt fix the call and retry.
|
||||
body_text = getattr(getattr(exc, "response", None), "text", "")
|
||||
if body_text:
|
||||
detail = f" — {body_text[:300]}"
|
||||
raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc
|
||||
pcm = np.frombuffer(response.content, dtype=np.int16)
|
||||
if pcm.size == 0:
|
||||
raise TtsError("ElevenLabs returned no dialogue audio")
|
||||
return pcm, config.TTS_SAMPLE_RATE
|
||||
|
||||
|
||||
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None) -> bool:
|
||||
"""Play a whole clip. Returns True if it finished, False if *should_stop*
|
||||
(barge-in) cut it short. *should_stop* is polled while audio plays — each
|
||||
@@ -134,11 +213,12 @@ def speak_offline(text: str) -> None:
|
||||
engine.runAndWait()
|
||||
|
||||
|
||||
def speak(text: str, on_error=None, should_stop=None) -> bool:
|
||||
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None) -> bool:
|
||||
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
|
||||
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
|
||||
exception when ElevenLabs fails (useful for logging) — a fallback still
|
||||
runs either way. Returns False if barge-in interrupted playback.
|
||||
*voice_id* overrides the configured voice for this line only.
|
||||
|
||||
The text is sanitized first (speech_text.for_speech): server replies are
|
||||
written for a chat window, and a voice reads markdown/emoji literally
|
||||
@@ -149,12 +229,16 @@ def speak(text: str, on_error=None, should_stop=None) -> bool:
|
||||
return True
|
||||
if config.TTS_STREAMING:
|
||||
try:
|
||||
return play_stream(stream_pcm(text), config.TTS_SAMPLE_RATE, should_stop=should_stop)
|
||||
return play_stream(
|
||||
stream_pcm(text, voice_id=voice_id),
|
||||
config.TTS_SAMPLE_RATE,
|
||||
should_stop=should_stop,
|
||||
)
|
||||
except TtsError as exc:
|
||||
if on_error is not None:
|
||||
on_error(exc)
|
||||
try:
|
||||
pcm, sample_rate = synthesize_pcm(text)
|
||||
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
|
||||
return play_pcm(pcm, sample_rate, should_stop=should_stop)
|
||||
except TtsError as exc:
|
||||
if on_error is not None:
|
||||
|
||||
+38
-1
@@ -66,9 +66,38 @@ DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3")
|
||||
ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "")
|
||||
ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "")
|
||||
ELEVENLABS_MODEL_ID = os.environ.get("ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
||||
# eleven_flash_v2 is English-only, and the two cases that swap the voice
|
||||
# (server-picked `speak_as`, or a reply with non-ASCII in it) are usually
|
||||
# exactly the cases where the reply isn't English — see tts.model_for().
|
||||
ELEVENLABS_MULTILINGUAL_MODEL_ID = os.environ.get(
|
||||
"ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5"
|
||||
)
|
||||
# ElevenLabs PCM output formats are named pcm_<sample_rate>.
|
||||
TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "24000"))
|
||||
|
||||
# Does a voice the server picks (its speak_as marker — "talk like a pirate",
|
||||
# "say that in Japanese") stay on for later replies, or last one reply only?
|
||||
# Sticky by default: the server tags a single reply and does *not* keep the
|
||||
# voice id in its history, so a one-reply-only voice can't be re-used when
|
||||
# you say "keep talking like that" — it would have to search for a voice
|
||||
# again. Reset it from the tray ("Use default voice") or by restarting.
|
||||
VOICE_STICKY = os.environ.get("VOICE_STICKY", "true").lower() in ("1", "true", "yes", "on")
|
||||
|
||||
# ── multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
|
||||
# Lets Bolt play a short scene in several voices with delivery tags the v3
|
||||
# model acts on ("[cheerfully] Hello"), instead of one voice reading a line.
|
||||
# Driven by the server through the `dialoguectl` relayed command — see
|
||||
# dialogue.py. Costs a separate (slower, whole-clip) request per scene, so
|
||||
# it's a set piece, not the normal reply path.
|
||||
#
|
||||
# DIALOGUE_VOICES names the cast: "narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD".
|
||||
# The name "self" always resolves to the voice the pet is currently using,
|
||||
# including one the server picked with speak_as.
|
||||
|
||||
DIALOGUE = os.environ.get("DIALOGUE", "true").lower() in ("1", "true", "yes", "on")
|
||||
DIALOGUE_MODEL_ID = os.environ.get("DIALOGUE_MODEL_ID", "eleven_v3")
|
||||
DIALOGUE_VOICES = os.environ.get("DIALOGUE_VOICES", "")
|
||||
|
||||
# ── mic / VAD (same tuning knobs as bolt_desk.py) ───────────────────────────
|
||||
|
||||
MIC_DEVICE = os.environ.get("MIC_DEVICE", "") or None # sounddevice name/index
|
||||
@@ -91,7 +120,7 @@ GRACE_SECONDS = float(os.environ.get("VAD_GRACE_SECONDS", "4"))
|
||||
# keeps feeding it noise. 0 means no cap.
|
||||
|
||||
FOLLOW_UP_LISTEN = os.environ.get("FOLLOW_UP_LISTEN", "true").lower() in ("1", "true", "yes", "on")
|
||||
FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "3"))
|
||||
FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "10"))
|
||||
FOLLOW_UP_GRACE_SECONDS = float(os.environ.get("FOLLOW_UP_GRACE_SECONDS", "7"))
|
||||
|
||||
COMMAND_TIMEOUT_SECONDS = int(os.environ.get("COMMAND_TIMEOUT_SECONDS", "30"))
|
||||
@@ -110,6 +139,14 @@ SUDO_ASKPASS_HELPER = os.environ.get("SUDO_ASKPASS_HELPER", "") # blank = auto-
|
||||
SUDO_COMMAND_TIMEOUT_SECONDS = int(os.environ.get("SUDO_COMMAND_TIMEOUT_SECONDS", "180"))
|
||||
HEARTBEAT_INTERVAL_SECONDS = float(os.environ.get("HEARTBEAT_INTERVAL_SECONDS", "60"))
|
||||
|
||||
# ── self-restart ────────────────────────────────────────────────────────────
|
||||
# `petctl self_restart` lets Bolt restart the pet after editing its code, so
|
||||
# he can see his own change running instead of waiting for someone to restart
|
||||
# it by hand. The code is import-checked in a subprocess first, and the reason
|
||||
# is carried across the restart so the new process can report back — see
|
||||
# self_restart.py. SELF_RESTART_MAX/_WINDOW_SECONDS bound the crash-loop case.
|
||||
SELF_RESTART = os.environ.get("SELF_RESTART", "true").lower() in ("1", "true", "yes", "on")
|
||||
|
||||
# ── barge-in (interrupt playback while the pet is talking) ──────────────────
|
||||
# The mic stays live while the pet talks. BARGE_IN_MODE decides what counts
|
||||
# as an interruption:
|
||||
|
||||
+211
-6
@@ -20,10 +20,12 @@ from typing import Optional
|
||||
from PySide6.QtCore import QObject, Signal
|
||||
|
||||
from . import (
|
||||
config, file_delivery, file_ops, history as history_mod,
|
||||
monitors as monitors_mod, notifications, pet_actions, quiet,
|
||||
screen_context, screen_text, server_client, speech_text, updater,
|
||||
config, dialogue as dialogue_mod, file_delivery, file_ops,
|
||||
history as history_mod, monitors as monitors_mod, notifications,
|
||||
pet_actions, quiet, screen_context, screen_text, self_restart,
|
||||
server_client, speech_text, updater,
|
||||
)
|
||||
from . import __version__
|
||||
from .audio import barge_in, mic, stt, tts, wake_word
|
||||
from .state import PetState, PetStateMachine
|
||||
|
||||
@@ -38,6 +40,7 @@ class PetController(QObject):
|
||||
log = Signal(str)
|
||||
action = Signal(dict) # parsed petctl action for the UI to perform
|
||||
napping = Signal(bool) # quiet hours / fullscreen do-not-disturb
|
||||
voice_changed = Signal(str) # name of the server-picked voice ("" = default)
|
||||
restart_requested = Signal(str) # version we just updated to
|
||||
finished = Signal()
|
||||
|
||||
@@ -66,6 +69,13 @@ class PetController(QObject):
|
||||
self._wake_threshold = config.WAKE_WORD_THRESHOLD
|
||||
self._near_misses = wake_word.NearMissLog()
|
||||
|
||||
# The voice the server last picked for us with `speak_as` ("" = the
|
||||
# configured default). Held here rather than passed straight through
|
||||
# to one tts.speak() call because it's sticky by default — see
|
||||
# _apply_voice for why.
|
||||
self._voice_id = ""
|
||||
self._voice_name = ""
|
||||
|
||||
self._barge_in: Optional[barge_in.BargeInDetector] = None
|
||||
self._napping = False
|
||||
self._nap_forced: Optional[bool] = None # petctl nap on/off overrides the schedule
|
||||
@@ -80,6 +90,9 @@ class PetController(QObject):
|
||||
|
||||
self._last_update_check = 0.0
|
||||
self._update_pending = False # applied on disk, waiting for the restart
|
||||
# Armed by `petctl self_restart`, fired after the turn it was asked in
|
||||
# (see _arm_self_restart for why it can't happen inline).
|
||||
self._restart_context = None
|
||||
|
||||
self._notification_watcher: Optional[notifications.NotificationWatcher] = None
|
||||
self._notification_gate = notifications.NotificationGate(
|
||||
@@ -117,6 +130,20 @@ class PetController(QObject):
|
||||
def reset_wake_stats(self) -> None:
|
||||
self._near_misses.clear()
|
||||
|
||||
def current_voice(self) -> str:
|
||||
"""Name (or id) of the server-picked voice in use, "" for the default."""
|
||||
return self._voice_name or self._voice_id
|
||||
|
||||
def reset_voice(self) -> None:
|
||||
"""Drop a server-picked voice and go back to Bolt's own. The tray's
|
||||
way out of a voice you didn't want to keep — the server has no way to
|
||||
ask for the default back, since it never learns what it is."""
|
||||
if not self._voice_id:
|
||||
return
|
||||
self._voice_id = self._voice_name = ""
|
||||
self.log.emit("Voice: back to the default.")
|
||||
self.voice_changed.emit("")
|
||||
|
||||
def stop(self) -> None:
|
||||
self._running = False
|
||||
self._talk_now.set() # wake up anything blocked waiting on it
|
||||
@@ -167,6 +194,7 @@ class PetController(QObject):
|
||||
except Exception as exc:
|
||||
self.log.emit(f"Server not reachable yet ({exc}) — will keep trying per-request.")
|
||||
self._start_notification_bridge()
|
||||
self._report_self_restart()
|
||||
self._loop()
|
||||
if self._notification_watcher is not None:
|
||||
self._notification_watcher.stop()
|
||||
@@ -260,8 +288,10 @@ class PetController(QObject):
|
||||
return
|
||||
|
||||
self._check_deliveries()
|
||||
self._speak(reply)
|
||||
self._apply_voice(reply)
|
||||
self._speak(reply.text)
|
||||
self._state.transition(PetState.IDLE)
|
||||
self._maybe_self_restart()
|
||||
|
||||
def _with_context(self, text: str) -> str:
|
||||
"""Everything the server gets alongside what you actually said: the
|
||||
@@ -307,6 +337,18 @@ class PetController(QObject):
|
||||
kind = action["action"]
|
||||
if kind == "monitors":
|
||||
return monitors_mod.describe(self._monitors, self._pet_monitor)
|
||||
if kind == "self_restart":
|
||||
return self._arm_self_restart(action.get("reason") or "")
|
||||
if kind == "voice":
|
||||
# Answered here, not by describe(): the UI has no part in it,
|
||||
# and the server needs to hear whether there was anything to
|
||||
# drop — it can't see which voice we're using.
|
||||
previous = self.current_voice()
|
||||
self.reset_voice()
|
||||
return (
|
||||
f"[pet] back to your own voice (was {previous})" if previous
|
||||
else "[pet] already using your own voice"
|
||||
)
|
||||
if kind == "read":
|
||||
return self._read_screen(action["target"])
|
||||
if kind == "jump":
|
||||
@@ -327,6 +369,14 @@ class PetController(QObject):
|
||||
self.action.emit(action)
|
||||
return pet_actions.describe(action)
|
||||
|
||||
try:
|
||||
scene = dialogue_mod.parse(command)
|
||||
except dialogue_mod.DialogueError as exc:
|
||||
self.log.emit(f"dialoguectl: {exc}")
|
||||
return f"[dialogue] {exc}"
|
||||
if scene is not None:
|
||||
return self._play_dialogue(scene)
|
||||
|
||||
try:
|
||||
file_action = file_ops.parse(command)
|
||||
except file_ops.FileOpError as exc:
|
||||
@@ -365,6 +415,159 @@ class PetController(QObject):
|
||||
self.log.emit(f"Reading monitor {monitor.number} ({monitor.name})…")
|
||||
return screen_text.read_monitor(monitor, limit)
|
||||
|
||||
def _arm_self_restart(self, reason: str) -> str:
|
||||
"""`petctl self_restart` — check the code, then arm a restart.
|
||||
|
||||
Nothing restarts here. The tool result has to get back up the relay
|
||||
before this process can die (otherwise the server waits out its
|
||||
timeout on a turn that will never finish), so the restart is armed and
|
||||
`_maybe_self_restart` fires it once the turn has been spoken. The
|
||||
preflight import runs *now*, in this turn, so a syntax error Bolt just
|
||||
introduced comes back as something he can read and fix rather than as
|
||||
a pet that never comes back."""
|
||||
if not config.SELF_RESTART:
|
||||
return "[pet] self-restart is disabled on this device (SELF_RESTART=false)"
|
||||
if self._restart_context is not None:
|
||||
return "[pet] a restart is already armed for the end of this turn"
|
||||
try:
|
||||
self_restart.check_loop_guard(self_restart.load())
|
||||
self.log.emit("Self-restart requested — checking the code imports first…")
|
||||
self_restart.preflight()
|
||||
except self_restart.RestartError as exc:
|
||||
self.log.emit(f"Self-restart refused: {exc}")
|
||||
return f"[pet] restart refused — {exc}"
|
||||
|
||||
recent = [entry.text[:120] for entry in self.history.entries()[-4:]]
|
||||
self._restart_context = self_restart.arm(
|
||||
reason or "no reason given",
|
||||
verify=reason,
|
||||
version=__version__,
|
||||
session=config.SESSION_ID,
|
||||
recent=recent,
|
||||
)
|
||||
self.log.emit("Self-restart armed; it happens after this turn.")
|
||||
return (
|
||||
"[pet] code imports cleanly; restarting as soon as this turn finishes. "
|
||||
"I'll come back and tell you what version I'm on and what I found — "
|
||||
"wrap up your reply now, the next thing you hear from me is the report."
|
||||
)
|
||||
|
||||
def _maybe_self_restart(self) -> bool:
|
||||
"""Fire an armed restart, once the turn is over and the reply spoken.
|
||||
|
||||
Returns True if a restart was requested, so the caller can stop
|
||||
driving the pipeline — the process is on its way out."""
|
||||
if self._restart_context is None:
|
||||
return False
|
||||
self._update_pending = True # same latch the updater uses: no double restart
|
||||
self.log.emit("Restarting now.")
|
||||
self._state.force(PetState.IDLE)
|
||||
self.restart_requested.emit(f"self-restart: {self._restart_context.reason[:60]}")
|
||||
return True
|
||||
|
||||
def _report_self_restart(self) -> None:
|
||||
"""On the way up: tell the server we're back, and why we left.
|
||||
|
||||
Runs once, before the listen loop starts, and only when a context file
|
||||
was left behind. The report goes through the ordinary conversation
|
||||
path, so Bolt's answer is spoken out loud like any other turn — which
|
||||
is what makes "restart and check the sprites load" finish as a
|
||||
sentence instead of a silence."""
|
||||
context = self_restart.load()
|
||||
if context is None:
|
||||
return
|
||||
self_restart.clear()
|
||||
message = self_restart.report(context, version=__version__)
|
||||
self.log.emit(f"Back from a self-restart ({context.reason[:80]}).")
|
||||
self.history.add(history_mod.SYSTEM, message, time.time())
|
||||
try:
|
||||
reply = server_client.converse(message, on_command=self._handle_command)
|
||||
except server_client.ServerError as exc:
|
||||
# The restart still worked; only the report failed. Say so locally
|
||||
# rather than pretending nothing happened.
|
||||
self.log.emit(f"Couldn't report the restart to the server: {exc}")
|
||||
return
|
||||
self._apply_voice(reply)
|
||||
if reply.text.strip() and not self._napping:
|
||||
self._speak(reply.text)
|
||||
self._state.force(PetState.IDLE)
|
||||
|
||||
def _play_dialogue(self, scene: dict) -> str:
|
||||
"""Play a `dialoguectl` scene and report back up the relay.
|
||||
|
||||
This runs *mid-turn* (the server is still waiting on the tool result),
|
||||
so the pet has to look like it's talking and then go back to waiting —
|
||||
hence the TALKING → THINKING leg rather than the usual return to IDLE.
|
||||
Everything a normal reply gets, a scene gets too: the bubble, the
|
||||
transcript, and barge-in, so a long scene can be talked over exactly
|
||||
like a long answer."""
|
||||
if not config.DIALOGUE:
|
||||
return "[dialogue] disabled on this device (DIALOGUE=false)"
|
||||
try:
|
||||
inputs = dialogue_mod.resolve(
|
||||
scene,
|
||||
voices=dialogue_mod.parse_voice_map(config.DIALOGUE_VOICES),
|
||||
self_voice=self._voice_id or config.ELEVENLABS_VOICE_ID,
|
||||
)
|
||||
except dialogue_mod.DialogueError as exc:
|
||||
self.log.emit(f"dialoguectl: {exc}")
|
||||
return f"[dialogue] {exc}"
|
||||
|
||||
text = dialogue_mod.spoken_text(scene)
|
||||
self.log.emit(f"Dialogue ({len(inputs)} lines): {text[:120]}")
|
||||
try:
|
||||
pcm, sample_rate = tts.synthesize_dialogue(
|
||||
inputs, model_id=scene.get("model"), stability=scene.get("stability")
|
||||
)
|
||||
except tts.TtsError as exc:
|
||||
self.log.emit(f"Dialogue failed: {exc}")
|
||||
# Reported, not raised: the server can read this, shorten the
|
||||
# scene or fix the voice, and try again inside the same turn.
|
||||
return f"[dialogue] couldn't synthesize it: {exc}"
|
||||
|
||||
resume = self._state.state
|
||||
self._state.transition(PetState.TALKING)
|
||||
self.said.emit(speech_text.for_display(text))
|
||||
self.history.add(history_mod.PET, text, time.time())
|
||||
|
||||
should_stop = None
|
||||
if self._barge_in is not None:
|
||||
self._barge_in.reset()
|
||||
should_stop = self._barge_in.check
|
||||
completed = tts.play_pcm(pcm, sample_rate, should_stop=should_stop)
|
||||
if self._barge_in is not None:
|
||||
self._barge_in.reset() # the pet's own voices are in the wake window
|
||||
if resume in (PetState.THINKING, PetState.IDLE):
|
||||
self._state.transition(resume)
|
||||
|
||||
if not completed:
|
||||
return dialogue_mod.describe(scene) + " (interrupted — they talked over it)"
|
||||
return dialogue_mod.describe(scene)
|
||||
|
||||
def _apply_voice(self, reply) -> None:
|
||||
"""Adopt (or drop) the voice the server tagged this reply with.
|
||||
|
||||
The server's `speak_as` marker names an ElevenLabs voice it just
|
||||
picked — and it adds a Voice Library pick to the account first, so by
|
||||
the time the id gets here it's usable for TTS. It tags *one* reply,
|
||||
but the voice sticks by default: the server strips the marker before
|
||||
storing the turn, so it can't recall the id later, and "keep talking
|
||||
like that" would otherwise send it searching for a voice all over
|
||||
again. `VOICE_STICKY=false` makes each pick last exactly one reply.
|
||||
|
||||
Untagged replies never *change* the voice — with stickiness on they
|
||||
just keep whatever's in use, which is what makes the rest of the
|
||||
conversation stay in the requested voice."""
|
||||
voice_id = getattr(reply, "voice_id", "")
|
||||
if voice_id:
|
||||
if voice_id != self._voice_id:
|
||||
self._voice_id = voice_id
|
||||
self._voice_name = getattr(reply, "voice_name", "") or ""
|
||||
self.log.emit(f"Voice: {self.current_voice()}")
|
||||
self.voice_changed.emit(self.current_voice())
|
||||
elif not config.VOICE_STICKY:
|
||||
self.reset_voice()
|
||||
|
||||
def _speak(self, text: str) -> None:
|
||||
self._state.transition(PetState.TALKING)
|
||||
# Bubble gets the markdown stripped but emoji kept (it can't render
|
||||
@@ -382,6 +585,7 @@ class PetController(QObject):
|
||||
text,
|
||||
on_error=lambda exc: self.log.emit(f"TTS failed: {exc}"),
|
||||
should_stop=should_stop,
|
||||
voice_id=self._voice_id or None,
|
||||
)
|
||||
# Read the scoring history *before* resetting, or the log reports the
|
||||
# blank counters instead of what actually fired.
|
||||
@@ -504,8 +708,9 @@ class PetController(QObject):
|
||||
self.log.emit(f"Couldn't forward notification: {exc}")
|
||||
return
|
||||
self._check_deliveries()
|
||||
if reply.strip():
|
||||
self._speak(reply)
|
||||
self._apply_voice(reply)
|
||||
if reply.text.strip():
|
||||
self._speak(reply.text)
|
||||
self._state.transition(PetState.IDLE)
|
||||
|
||||
# ── file delivery ────────────────────────────────────────────────────
|
||||
|
||||
@@ -0,0 +1,219 @@
|
||||
"""`dialoguectl` — multi-voice dialogue playback (ElevenLabs Text to Dialogue).
|
||||
|
||||
Normal replies are one voice saying one thing (audio/tts.py). This is the
|
||||
other mode: a short *scene* — two or more voices, with delivery tags the v3
|
||||
model acts on (`[cheerfully]`, `[stuttering]`, `[whispering]`) — synthesized
|
||||
as a single take so the timing and reactions between lines actually sound
|
||||
like a conversation rather than clips glued together.
|
||||
|
||||
Wire format, the same discipline as file_ops.py and for the same reason: it
|
||||
rides the server's ordinary `command` tool marker, whose extractor only
|
||||
captures up to the next newline, so the payload is a **single-line compact
|
||||
JSON object**.
|
||||
|
||||
dialoguectl {"lines": [{"voice": "self", "text": "[cheerfully] Morning!"},
|
||||
{"voice": "narrator", "text": "[whispering] He lies."}]}
|
||||
|
||||
The ElevenLabs field names are accepted too (`inputs` / `voice_id`), because
|
||||
the model has read that API and copying its shape is the obvious thing to
|
||||
try:
|
||||
|
||||
dialoguectl {"inputs": [{"voice_id": "9BWtsMINqrJLrRacOk9x", "text": "hi"}]}
|
||||
|
||||
Voices are *named*, not pasted as ids. `DIALOGUE_VOICES` in .env maps names
|
||||
to ids (`narrator:9BWts…,villain:IKne3…`), and `self` always means the voice
|
||||
the pet is speaking with right now — including a voice the server picked
|
||||
mid-conversation with `speak_as`, so a scene featuring Bolt sounds like
|
||||
whoever Bolt currently is.
|
||||
|
||||
Pure parsing and validation here; the HTTP call is
|
||||
`audio/tts.synthesize_dialogue` and the playback/state handling is
|
||||
`controller._play_dialogue`, matching the parse/execute split used by
|
||||
pet_actions.py and file_ops.py.
|
||||
|
||||
The API's own limits are enforced *here*, before the request goes out, so a
|
||||
mistake comes back through the tool-result relay as a sentence Bolt can act
|
||||
on ("too many characters, split it") rather than as an HTTP 422 he can't see.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from typing import Iterable, Optional
|
||||
|
||||
_PREFIXES = ("dialoguectl", "dialogue", "scene")
|
||||
|
||||
# ElevenLabs Text to Dialogue limits (docs, 2026-07): at most 10 distinct
|
||||
# voice ids per request and ~2000 characters across all inputs.
|
||||
MAX_VOICES = 10
|
||||
MAX_CHARS = 2000
|
||||
|
||||
# Names that always mean "the voice the pet is using right now".
|
||||
SELF_NAMES = ("self", "bolt", "me", "pet")
|
||||
|
||||
# A raw ElevenLabs voice id: 20 URL-safe characters, no separators. Used to
|
||||
# tell "the model pasted an id" from "the model used a name".
|
||||
_VOICE_ID_RE = re.compile(r"^[A-Za-z0-9]{20}$")
|
||||
|
||||
|
||||
class DialogueError(Exception):
|
||||
"""Bad dialoguectl syntax or an unusable request — reported back to the
|
||||
server as this command's output."""
|
||||
|
||||
|
||||
def is_dialogue_command(command: str) -> bool:
|
||||
parts = (command or "").strip().split(None, 1)
|
||||
return bool(parts) and parts[0].lower() in _PREFIXES
|
||||
|
||||
|
||||
def parse(command: str) -> Optional[dict]:
|
||||
"""Parse `dialoguectl <json>` into {"lines": [{"voice", "text"}], ...}.
|
||||
|
||||
Returns None if this isn't a dialogue command at all (the caller then
|
||||
tries filectl, then a real shell command). Raises DialogueError on a
|
||||
dialogue command that doesn't make sense."""
|
||||
if not is_dialogue_command(command):
|
||||
return None
|
||||
_, _, payload = (command or "").strip().partition(" ")
|
||||
payload = payload.strip()
|
||||
if not payload:
|
||||
raise DialogueError(
|
||||
'dialoguectl needs a JSON argument, e.g. dialoguectl {"lines": '
|
||||
'[{"voice": "self", "text": "[cheerfully] hello"}]}'
|
||||
)
|
||||
try:
|
||||
data = json.loads(payload)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise DialogueError(
|
||||
f"couldn't parse the JSON ({exc}). It must be one line of compact "
|
||||
"JSON — put line breaks inside text as \\n, never as real newlines."
|
||||
) from exc
|
||||
if not isinstance(data, dict):
|
||||
raise DialogueError("the argument must be a JSON object, not a list or a bare value")
|
||||
|
||||
raw_lines = data.get("lines")
|
||||
if raw_lines is None:
|
||||
raw_lines = data.get("inputs") # the ElevenLabs field name
|
||||
if not isinstance(raw_lines, list) or not raw_lines:
|
||||
raise DialogueError('needs a non-empty "lines" array of {"voice", "text"} objects')
|
||||
|
||||
lines: list[dict] = []
|
||||
for index, entry in enumerate(raw_lines, start=1):
|
||||
if not isinstance(entry, dict):
|
||||
raise DialogueError(f"line {index} must be an object with 'voice' and 'text'")
|
||||
text = str(entry.get("text") or "").strip()
|
||||
if not text:
|
||||
raise DialogueError(f"line {index} has no text")
|
||||
voice = str(entry.get("voice") or entry.get("voice_id") or "self").strip()
|
||||
lines.append({"voice": voice, "text": text})
|
||||
|
||||
action = {"action": "dialogue", "lines": lines}
|
||||
model = str(data.get("model") or data.get("model_id") or "").strip()
|
||||
if model:
|
||||
action["model"] = model
|
||||
stability = data.get("stability")
|
||||
if stability is not None:
|
||||
try:
|
||||
action["stability"] = min(1.0, max(0.0, float(stability)))
|
||||
except (TypeError, ValueError):
|
||||
raise DialogueError("stability must be a number between 0 and 1") from None
|
||||
return action
|
||||
|
||||
|
||||
def parse_voice_map(spec: str) -> dict[str, str]:
|
||||
"""Parse DIALOGUE_VOICES ("narrator:9BWts…, villain:IKne3…") into a map.
|
||||
|
||||
Malformed entries are skipped rather than raising: a typo in .env should
|
||||
cost that one voice, not the whole feature."""
|
||||
voices: dict[str, str] = {}
|
||||
for chunk in str(spec or "").split(","):
|
||||
name, separator, voice_id = chunk.partition(":")
|
||||
name, voice_id = name.strip().lower(), voice_id.strip()
|
||||
if separator and name and voice_id:
|
||||
voices[name] = voice_id
|
||||
return voices
|
||||
|
||||
|
||||
def resolve(
|
||||
action: dict,
|
||||
*,
|
||||
voices: Optional[dict] = None,
|
||||
self_voice: str = "",
|
||||
) -> list[dict]:
|
||||
"""Turn parsed lines into the API's `inputs`, resolving names to ids.
|
||||
|
||||
*self_voice* is the pet's current voice (which may be a `speak_as` pick,
|
||||
not the configured default), so "self" tracks whoever Bolt sounds like
|
||||
right now."""
|
||||
known = dict(voices or {})
|
||||
resolved: list[dict] = []
|
||||
for index, line in enumerate(action.get("lines") or [], start=1):
|
||||
name = str(line.get("voice") or "self")
|
||||
key = name.lower()
|
||||
if key in SELF_NAMES:
|
||||
voice_id = self_voice
|
||||
if not voice_id:
|
||||
raise DialogueError(
|
||||
"no voice is configured for the pet itself — set "
|
||||
"ELEVENLABS_VOICE_ID, or name a voice from DIALOGUE_VOICES"
|
||||
)
|
||||
elif key in known:
|
||||
voice_id = known[key]
|
||||
elif _VOICE_ID_RE.match(name):
|
||||
voice_id = name # a raw id pasted straight from the voice library
|
||||
else:
|
||||
available = ", ".join(sorted(known) + list(SELF_NAMES[:1])) or "self"
|
||||
raise DialogueError(
|
||||
f"line {index}: unknown voice {name!r}. Known names: {available}. "
|
||||
"Use one of those, 'self' for your own voice, or a raw voice id."
|
||||
)
|
||||
resolved.append({"text": str(line.get("text") or ""), "voice_id": voice_id})
|
||||
|
||||
check_limits(resolved)
|
||||
return resolved
|
||||
|
||||
|
||||
def check_limits(inputs: Iterable[dict], *, max_voices: int = MAX_VOICES,
|
||||
max_chars: int = MAX_CHARS) -> None:
|
||||
"""Enforce the API's own limits before spending a request on a 422."""
|
||||
entries = list(inputs)
|
||||
if not entries:
|
||||
raise DialogueError("no lines to speak")
|
||||
distinct = {entry["voice_id"] for entry in entries}
|
||||
if len(distinct) > max_voices:
|
||||
raise DialogueError(
|
||||
f"{len(distinct)} different voices — the limit is {max_voices} per scene"
|
||||
)
|
||||
total = sum(len(entry["text"]) for entry in entries)
|
||||
if total > max_chars:
|
||||
raise DialogueError(
|
||||
f"{total} characters — the limit is {max_chars} per scene. "
|
||||
"Split it into two dialoguectl calls."
|
||||
)
|
||||
|
||||
|
||||
def spoken_text(action: dict) -> str:
|
||||
"""The scene as readable text, for the speech bubble and the transcript.
|
||||
|
||||
Delivery tags are stripped: `[cheerfully]` is a stage direction for the
|
||||
model, not something to show (or, via tts.speak's sanitizer, to read out)."""
|
||||
parts = []
|
||||
for line in action.get("lines") or []:
|
||||
text = re.sub(r"\[[^\]]{1,40}\]", " ", str(line.get("text") or ""))
|
||||
text = " ".join(text.split())
|
||||
if text:
|
||||
parts.append(text)
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def describe(action: dict, *, played: bool = True) -> str:
|
||||
"""The tool-result string handed back to the server."""
|
||||
lines = action.get("lines") or []
|
||||
voices = sorted({str(line.get("voice") or "self") for line in lines})
|
||||
if not played:
|
||||
return f"[dialogue] not played ({len(lines)} lines)"
|
||||
return (
|
||||
f"[dialogue] played {len(lines)} line{'s' if len(lines) != 1 else ''} "
|
||||
f"in {len(voices)} voice{'s' if len(voices) != 1 else ''}: {', '.join(voices)}"
|
||||
)
|
||||
+25
-1
@@ -44,9 +44,17 @@ HELP = (
|
||||
"petctl emote <" + "|".join(EMOTES) + ">\n"
|
||||
"petctl say <text>\n"
|
||||
"petctl wander on|off\n"
|
||||
"petctl nap on|off"
|
||||
"petctl nap on|off\n"
|
||||
"petctl voice reset\n"
|
||||
"petctl self_restart [why]"
|
||||
)
|
||||
|
||||
# `petctl voice` only ever goes one way: back to the configured voice. Picking
|
||||
# a *different* one is the server's job (its speak_as reply marker), and it
|
||||
# already knows how — what it has no way to say is "never mind, be yourself
|
||||
# again", because it was never told which voice that is.
|
||||
VOICE_RESETS = ("reset", "default", "normal", "own", "back", "mine", "yours")
|
||||
|
||||
|
||||
class ActionError(Exception):
|
||||
"""Bad petctl syntax — reported back to the server as command output."""
|
||||
@@ -129,6 +137,22 @@ def parse(command: str) -> Optional[dict]:
|
||||
raise ActionError("wander needs on or off")
|
||||
return {"action": "wander", "enabled": _bool_arg(args[0])}
|
||||
|
||||
if verb == "voice":
|
||||
target = (args[0].lower() if args else "reset").lstrip("-")
|
||||
if target not in VOICE_RESETS:
|
||||
raise ActionError(
|
||||
f"can't set a voice from petctl (got {args[0]!r}); "
|
||||
"use the speak_as reply marker to pick one. "
|
||||
"petctl voice reset goes back to the default voice."
|
||||
)
|
||||
return {"action": "voice", "voice": "default"}
|
||||
|
||||
if verb in ("self_restart", "restart", "reboot"):
|
||||
# Free text, not a fixed grammar: the argument is a note to the pet's
|
||||
# *next* process about why it died and what to look at when it comes
|
||||
# back, so anything the model wants to tell future-itself is valid.
|
||||
return {"action": "self_restart", "reason": " ".join(args).strip()}
|
||||
|
||||
if verb in ("nap", "sleep", "dnd"):
|
||||
if not args:
|
||||
raise ActionError("nap needs on or off")
|
||||
|
||||
@@ -0,0 +1,235 @@
|
||||
"""`petctl self_restart` — the pet restarting itself, and remembering why.
|
||||
|
||||
Bolt can already edit this repo through `filectl` and run commands through the
|
||||
shell relay, which means he can change the pet's own code. What he could not
|
||||
do is *see the result*: the running process keeps the old modules in memory,
|
||||
so an edit is invisible until somebody restarts the pet by hand, and by then
|
||||
the conversation that motivated it is over. That makes the edit-test-review
|
||||
loop a human errand.
|
||||
|
||||
This closes the loop. The tricky part is that the thing being asked to report
|
||||
back is the thing that dies, so the mechanism is built around three problems:
|
||||
|
||||
1. **The turn must survive.** A restart mid-turn would kill the HTTP tool
|
||||
relay before the result was posted, and the server would sit waiting until
|
||||
it timed out — the conversation lost, with no explanation. So the command
|
||||
only *arms* the restart: it returns immediately, the turn finishes and Bolt
|
||||
speaks his reply, and the restart happens after (see
|
||||
`controller._maybe_self_restart`), exactly like the updater's "only between
|
||||
turns" rule.
|
||||
2. **A broken edit must not be fatal.** Before anything is armed, the new code
|
||||
is imported in a *subprocess* (`preflight`) — this process still holds the
|
||||
old modules, so importing here would prove nothing. A syntax error comes
|
||||
back as the command's output, in the same turn, and nothing restarts. That
|
||||
is the difference between "Bolt broke the pet and lost his own way to fix
|
||||
it" and "Bolt got a traceback and tried again".
|
||||
3. **The reason must outlive the process.** The context (why, what to check,
|
||||
which version, when) is written to disk before exec and read on the way
|
||||
back up, so the new process can open with "I'm back — you asked me to check
|
||||
X" instead of amnesia. That report goes to the server as a normal turn, so
|
||||
Bolt sees the result of his own change and can carry on.
|
||||
|
||||
A loop guard bounds the worst case: `MAX_RESTARTS` inside `WINDOW_SECONDS`
|
||||
and further self-restarts are refused with a reason, so an edit-restart-crash
|
||||
cycle stops on its own rather than spinning the process forever.
|
||||
|
||||
Pure-ish and injectable throughout (paths, clock, subprocess runner) so the
|
||||
whole thing is testable without ever restarting anything.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from . import config
|
||||
|
||||
# Lives in the cache dir, not the repo: it is transient state about *this*
|
||||
# machine's process, and it must never end up in a git diff of the checkout
|
||||
# Bolt is editing.
|
||||
DEFAULT_STATE_PATH = Path.home() / ".cache" / "bolt-pet" / "restart_context.json"
|
||||
|
||||
# Loop guard. Deliberately small: a healthy edit-check cycle is one restart
|
||||
# per change, and anything hammering past this is a crash loop, not work.
|
||||
MAX_RESTARTS = int(os.environ.get("SELF_RESTART_MAX", "5"))
|
||||
WINDOW_SECONDS = float(os.environ.get("SELF_RESTART_WINDOW_SECONDS", "900"))
|
||||
|
||||
# What the preflight subprocess imports. `ui.app` pulls in the widest slice of
|
||||
# the package (Qt, controller, audio, every helper), so if this imports, a
|
||||
# restart will at least reach the event loop.
|
||||
_PREFLIGHT_IMPORT = "import bolt_pet, bolt_pet.controller, bolt_pet.ui.app"
|
||||
|
||||
|
||||
class RestartError(Exception):
|
||||
"""A refused restart — reported back to the server as command output."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class RestartContext:
|
||||
"""What the dying process wants the next one to know."""
|
||||
|
||||
reason: str = ""
|
||||
verify: str = ""
|
||||
armed_at: float = 0.0
|
||||
version: str = ""
|
||||
session: str = ""
|
||||
recent: list = field(default_factory=list)
|
||||
restarts: list = field(default_factory=list) # timestamps, for the loop guard
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
def _now() -> float:
|
||||
return time.time()
|
||||
|
||||
|
||||
def load(path: Optional[Path] = None) -> Optional[RestartContext]:
|
||||
"""Read the context left by a previous process, or None."""
|
||||
target = Path(path or DEFAULT_STATE_PATH)
|
||||
try:
|
||||
data = json.loads(target.read_text(encoding="utf-8"))
|
||||
except (FileNotFoundError, json.JSONDecodeError, OSError):
|
||||
return None
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
known = {field_name for field_name in RestartContext().as_dict()}
|
||||
return RestartContext(**{k: v for k, v in data.items() if k in known})
|
||||
|
||||
|
||||
def save(context: RestartContext, path: Optional[Path] = None) -> None:
|
||||
"""Persist the context atomically — a half-written file on the way out
|
||||
would make the next process start confused instead of oriented."""
|
||||
target = Path(path or DEFAULT_STATE_PATH)
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
descriptor, temp_path = tempfile.mkstemp(dir=target.parent, prefix=".restart_", suffix=".tmp")
|
||||
try:
|
||||
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
|
||||
json.dump(context.as_dict(), handle, ensure_ascii=False, indent=1)
|
||||
os.replace(temp_path, target)
|
||||
except BaseException:
|
||||
try:
|
||||
os.unlink(temp_path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
|
||||
|
||||
def clear(path: Optional[Path] = None) -> None:
|
||||
"""Consume the context. Called once it has been reported, so the pet
|
||||
doesn't announce the same restart every time it starts."""
|
||||
try:
|
||||
Path(path or DEFAULT_STATE_PATH).unlink()
|
||||
except (FileNotFoundError, OSError):
|
||||
pass
|
||||
|
||||
|
||||
def recent_restarts(context: Optional[RestartContext], *, now: Optional[float] = None) -> list:
|
||||
current = now if now is not None else _now()
|
||||
stamps = list((context.restarts if context else []) or [])
|
||||
return [stamp for stamp in stamps if current - float(stamp) <= WINDOW_SECONDS]
|
||||
|
||||
|
||||
def check_loop_guard(context: Optional[RestartContext], *, now: Optional[float] = None) -> None:
|
||||
"""Refuse to restart if we've already done it too many times recently."""
|
||||
stamps = recent_restarts(context, now=now)
|
||||
if len(stamps) >= MAX_RESTARTS:
|
||||
raise RestartError(
|
||||
f"refusing: {len(stamps)} self-restarts in the last "
|
||||
f"{int(WINDOW_SECONDS / 60)} minutes. Something is looping — fix the "
|
||||
"cause, or wait for the window to clear before trying again."
|
||||
)
|
||||
|
||||
|
||||
def preflight(
|
||||
repo: Optional[Path] = None,
|
||||
run: Optional[Callable[..., Any]] = None,
|
||||
timeout: float = 120.0,
|
||||
) -> None:
|
||||
"""Import the current source in a subprocess; raise if it's broken.
|
||||
|
||||
This process has the *old* modules loaded, so importing in-process would
|
||||
happily succeed on a file that no longer parses. Mirrors
|
||||
`updater._smoke_test`, and exists for the same reason: never hand the
|
||||
session to code that can't start."""
|
||||
runner = run or subprocess.run
|
||||
root = Path(repo or config.HERE)
|
||||
try:
|
||||
completed = runner(
|
||||
[sys.executable, "-c", _PREFLIGHT_IMPORT],
|
||||
cwd=str(root), capture_output=True, text=True, timeout=timeout,
|
||||
env={**os.environ, "QT_QPA_PLATFORM": "offscreen"}, # no display needed to import
|
||||
)
|
||||
except Exception as exc: # subprocess itself failed to run
|
||||
raise RestartError(f"couldn't run the preflight import check: {exc}") from exc
|
||||
if completed.returncode != 0:
|
||||
detail = (completed.stderr or completed.stdout or "").strip()
|
||||
raise RestartError(
|
||||
"the current code does not import, so restarting would leave you with "
|
||||
f"nothing running. Fix this first:\n{detail[-800:]}"
|
||||
)
|
||||
|
||||
|
||||
def arm(
|
||||
reason: str,
|
||||
*,
|
||||
verify: str = "",
|
||||
version: str = "",
|
||||
session: str = "",
|
||||
recent: Optional[list] = None,
|
||||
path: Optional[Path] = None,
|
||||
now: Optional[float] = None,
|
||||
) -> RestartContext:
|
||||
"""Record why we're about to die, carrying the restart history forward."""
|
||||
current = now if now is not None else _now()
|
||||
previous = load(path)
|
||||
context = RestartContext(
|
||||
reason=" ".join(str(reason or "").split())[:400],
|
||||
verify=" ".join(str(verify or "").split())[:400],
|
||||
armed_at=current,
|
||||
version=str(version or ""),
|
||||
session=str(session or ""),
|
||||
recent=list(recent or [])[-6:],
|
||||
restarts=recent_restarts(previous, now=current) + [current],
|
||||
)
|
||||
save(context, path)
|
||||
return context
|
||||
|
||||
|
||||
def report(
|
||||
context: RestartContext,
|
||||
*,
|
||||
version: str = "",
|
||||
now: Optional[float] = None,
|
||||
) -> str:
|
||||
"""The message the new process sends the server on the way up.
|
||||
|
||||
Phrased as Bolt reporting to himself, because that is what it is: the
|
||||
server sees it as an ordinary turn, and the reply comes back through the
|
||||
normal pipeline — which is what lets "restart and check X" finish as a
|
||||
sentence spoken out loud."""
|
||||
current = now if now is not None else _now()
|
||||
took = max(0.0, current - float(context.armed_at or current))
|
||||
lines = [
|
||||
"[pet self-restart] I restarted myself and I'm back up.",
|
||||
f"- reason: {context.reason or 'not recorded'}",
|
||||
f"- took: {took:.1f}s",
|
||||
f"- version now running: {version or 'unknown'}"
|
||||
+ (f" (was {context.version})" if context.version and context.version != version else ""),
|
||||
]
|
||||
if context.verify:
|
||||
lines.append(f"- you wanted to check: {context.verify}")
|
||||
if context.recent:
|
||||
lines.append("- what we were doing before: " + " | ".join(str(x)[:120] for x in context.recent))
|
||||
lines.append(
|
||||
"The new code is loaded and running. If you wanted to verify something, "
|
||||
"check it now (filectl to read, command to test) and tell the user what you found."
|
||||
)
|
||||
return "\n".join(lines)
|
||||
@@ -9,6 +9,11 @@ memory, tools, and persona as Discord chat and the Linux voice client:
|
||||
... -> POST /desk/tool_result (repeat until the server sends a reply)
|
||||
reply <- returned to caller
|
||||
|
||||
A reply can also carry a voice (`voice_id`/`voice_name`), which is how the
|
||||
server's `speak_as` marker reaches us: Bolt searched the ElevenLabs voice
|
||||
library, picked one, and tagged the reply with it — the client is what
|
||||
actually speaks in it. See `Reply` and controller._apply_voice.
|
||||
|
||||
Kept dependency-free beyond `requests` so it's easy to unit test with mocks.
|
||||
"""
|
||||
|
||||
@@ -16,7 +21,7 @@ from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Callable, Optional
|
||||
from typing import Callable, NamedTuple, Optional
|
||||
|
||||
import requests
|
||||
|
||||
@@ -29,6 +34,17 @@ class ServerError(Exception):
|
||||
"""Raised when the server responds with an error payload or unreachable."""
|
||||
|
||||
|
||||
class Reply(NamedTuple):
|
||||
"""One final reply from the desk API. *voice_id* is set only when the
|
||||
server tagged this reply with a `speak_as` voice; *voice_name* is the
|
||||
human-readable name that came with it (may be empty even when the id
|
||||
isn't). Both empty means "say it in the usual voice"."""
|
||||
|
||||
text: str
|
||||
voice_id: str = ""
|
||||
voice_name: str = ""
|
||||
|
||||
|
||||
def _headers() -> dict:
|
||||
return {"X-Desk-Api-Key": config.API_KEY}
|
||||
|
||||
@@ -74,7 +90,7 @@ def converse(
|
||||
text: str,
|
||||
on_command: Callable[[str], str] = run_local_command,
|
||||
timeout: float = 120.0,
|
||||
) -> str:
|
||||
) -> Reply:
|
||||
"""Send one turn of conversation to the desk API, relaying any commands
|
||||
the server sends back until it produces a final reply.
|
||||
|
||||
@@ -111,7 +127,11 @@ def converse(
|
||||
raise ServerError(f"couldn't reach the server during tool relay: {exc}") from exc
|
||||
|
||||
if payload.get("type") == "reply":
|
||||
return str(payload.get("text") or "")
|
||||
return Reply(
|
||||
text=str(payload.get("text") or ""),
|
||||
voice_id=str(payload.get("voice_id") or ""),
|
||||
voice_name=str(payload.get("voice_name") or ""),
|
||||
)
|
||||
raise ServerError(str(payload.get("error") or "unknown server response"))
|
||||
|
||||
|
||||
|
||||
+6
-1
@@ -26,11 +26,16 @@ class PetState(str, Enum):
|
||||
# make the pet speak unprompted — a reminder firing, a nudge from the server
|
||||
# — without the user having said anything first, so there's no preceding
|
||||
# LISTENING/THINKING leg for that turn.
|
||||
#
|
||||
# TALKING -> THINKING is the mirror case: a `dialoguectl` scene is played
|
||||
# *mid-turn*, while the server is still waiting on the tool result, so the pet
|
||||
# talks and then goes back to waiting rather than falling to IDLE (which would
|
||||
# make it look like the turn had ended).
|
||||
_TRANSITIONS: dict[PetState, set[PetState]] = {
|
||||
PetState.IDLE: {PetState.LISTENING, PetState.TALKING, PetState.ERROR},
|
||||
PetState.LISTENING: {PetState.THINKING, PetState.IDLE, PetState.ERROR},
|
||||
PetState.THINKING: {PetState.TALKING, PetState.IDLE, PetState.ERROR},
|
||||
PetState.TALKING: {PetState.IDLE, PetState.ERROR},
|
||||
PetState.TALKING: {PetState.IDLE, PetState.THINKING, PetState.ERROR},
|
||||
PetState.ERROR: {PetState.IDLE},
|
||||
}
|
||||
|
||||
|
||||
@@ -80,6 +80,7 @@ def run() -> int:
|
||||
on_set_nap=_set_nap,
|
||||
on_show_history=history_window.show_refreshed,
|
||||
on_show_wake_tuner=tuner_window.show_refreshed,
|
||||
on_reset_voice=controller.reset_voice,
|
||||
)
|
||||
|
||||
def _handle_napping(napping: bool) -> None:
|
||||
@@ -87,6 +88,9 @@ def run() -> int:
|
||||
tray.set_napping(napping)
|
||||
|
||||
controller.napping.connect(_handle_napping)
|
||||
# The server can hand Bolt a different voice mid-conversation (speak_as);
|
||||
# the tray is where you get his own back.
|
||||
controller.voice_changed.connect(tray.set_voice)
|
||||
|
||||
# Push-to-talk: a global hook, because the pet window never has focus.
|
||||
# request_talk_now() only sets a threading.Event, so it's safe to call
|
||||
|
||||
+23
-1
@@ -1,6 +1,7 @@
|
||||
"""System tray icon — the pet window is frameless with no taskbar entry, so
|
||||
this menu is the only always-available way to control or exit it: talk now,
|
||||
mute, wander, click-through, nap, history, wake-word tuning, quit.
|
||||
mute, wander, click-through, nap, history, wake-word tuning, voice reset,
|
||||
quit.
|
||||
|
||||
Every entry is a plain callback passed in by ui/app.py; this file knows
|
||||
nothing about the controller or the pet window.
|
||||
@@ -46,6 +47,7 @@ class PetTray(QSystemTrayIcon):
|
||||
on_set_nap: Optional[Callable[[bool], None]] = None,
|
||||
on_show_history: Optional[Callable[[], None]] = None,
|
||||
on_show_wake_tuner: Optional[Callable[[], None]] = None,
|
||||
on_reset_voice: Optional[Callable[[], None]] = None,
|
||||
parent=None,
|
||||
):
|
||||
super().__init__(_make_icon(muted=False), parent)
|
||||
@@ -102,6 +104,16 @@ class PetTray(QSystemTrayIcon):
|
||||
tuner_action.triggered.connect(on_show_wake_tuner)
|
||||
menu.addAction(tuner_action)
|
||||
|
||||
# Only ever enabled while a server-picked voice (speak_as) is in use —
|
||||
# it's the way back from "talk like a pirate", which nothing else
|
||||
# undoes short of a restart.
|
||||
self._voice_action = None
|
||||
if on_reset_voice is not None:
|
||||
self._voice_action = QAction("Use default voice", menu)
|
||||
self._voice_action.setEnabled(False)
|
||||
self._voice_action.triggered.connect(on_reset_voice)
|
||||
menu.addAction(self._voice_action)
|
||||
|
||||
menu.addSeparator()
|
||||
quit_action = QAction("Quit", menu)
|
||||
quit_action.triggered.connect(on_quit)
|
||||
@@ -115,6 +127,16 @@ class PetTray(QSystemTrayIcon):
|
||||
self._mute_action.setChecked(self._muted)
|
||||
self._refresh_icon()
|
||||
|
||||
def set_voice(self, voice: str) -> None:
|
||||
"""Reflect the voice the controller is speaking in — a name (or id)
|
||||
when the server picked one, "" for Bolt's own."""
|
||||
if self._voice_action is None:
|
||||
return
|
||||
self._voice_action.setEnabled(bool(voice))
|
||||
self._voice_action.setText(
|
||||
f"Use default voice (now: {voice})" if voice else "Use default voice"
|
||||
)
|
||||
|
||||
def set_napping(self, napping: bool) -> None:
|
||||
"""Reflect a nap the *controller* decided on (quiet hours, fullscreen,
|
||||
or a petctl command) — not just ones clicked here."""
|
||||
|
||||
@@ -41,10 +41,10 @@ def test_full_turn_happy_path(monkeypatch, ctrl):
|
||||
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's the weather")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: "sunny and 72")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: controller_mod.server_client.Reply("sunny and 72"))
|
||||
spoken = []
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
@@ -150,7 +150,7 @@ def test_heartbeat_speaks_a_pending_announcement_when_idle(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: "don't forget your 3pm")
|
||||
spoken = []
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
|
||||
said = _capture(ctrl.said)
|
||||
|
||||
ctrl._maybe_heartbeat()
|
||||
|
||||
@@ -15,6 +15,7 @@ from PySide6.QtWidgets import QApplication
|
||||
|
||||
from bolt_pet import controller as controller_mod
|
||||
from bolt_pet.notifications import Notification
|
||||
from bolt_pet.server_client import Reply
|
||||
from bolt_pet.state import PetState
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
@@ -159,7 +160,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch
|
||||
ctrl._barge_in = detector
|
||||
logs = _capture(ctrl.log)
|
||||
|
||||
def interrupted_playback(text, on_error=None, should_stop=None):
|
||||
def interrupted_playback(text, on_error=None, should_stop=None, voice_id=None):
|
||||
# What really happens: frames get scored during playback, then one
|
||||
# clears the threshold and playback aborts.
|
||||
detector._frames, detector._peak, detector._last = 7, 0.81, 0.81
|
||||
@@ -179,7 +180,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch
|
||||
def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
|
||||
logs = _capture(ctrl.log)
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: False) # interrupted
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: False) # interrupted
|
||||
|
||||
ctrl._speak("a very long explanation")
|
||||
|
||||
@@ -189,7 +190,7 @@ def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
|
||||
|
||||
def test_uninterrupted_playback_does_not_queue_a_turn(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: True)
|
||||
ctrl._speak("short answer")
|
||||
assert not ctrl._talk_now.is_set()
|
||||
|
||||
@@ -201,7 +202,7 @@ def spoke(monkeypatch):
|
||||
"""Playback that always completes, so only the follow-up rule decides
|
||||
whether another turn is queued."""
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: True)
|
||||
|
||||
|
||||
def test_a_reply_ending_in_a_question_keeps_listening(spoke, ctrl):
|
||||
@@ -289,7 +290,7 @@ def test_follow_up_can_be_turned_off(spoke, monkeypatch, ctrl):
|
||||
|
||||
def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: False)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: False)
|
||||
ctrl._follow_ups = 3
|
||||
|
||||
ctrl._speak("a very long explanation")
|
||||
@@ -300,7 +301,7 @@ def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
|
||||
|
||||
def test_speech_is_recorded_in_the_history(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: True)
|
||||
ctrl._speak("**bold** reply")
|
||||
assert ctrl.history.last().text == "**bold** reply" # raw, for copy/paste
|
||||
|
||||
@@ -314,10 +315,10 @@ def test_the_active_window_rides_along_with_the_utterance(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.screen_context, "context_for",
|
||||
lambda text: f"{text}\n\n[on screen right now: app.py]")
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: True)
|
||||
sent = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: sent.append(text) or "that's a KeyError")
|
||||
lambda text, on_command=None: sent.append(text) or Reply("that's a KeyError"))
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
@@ -370,10 +371,10 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
|
||||
lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "you awake?")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: "always")
|
||||
lambda text, on_command=None: Reply("always"))
|
||||
spoken = []
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
|
||||
ctrl.set_napping(True)
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
@@ -388,9 +389,9 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
|
||||
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
|
||||
sent, spoken = [], []
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: sent.append(text) or "your build is green")
|
||||
lambda text, on_command=None: sent.append(text) or Reply("your build is green"))
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
|
||||
|
||||
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
|
||||
ctrl._drain_notifications()
|
||||
@@ -506,9 +507,9 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_
|
||||
lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "send me that file")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: "it's on the way")
|
||||
lambda text, on_command=None: Reply("it's on the way"))
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None: True)
|
||||
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
|
||||
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files",
|
||||
lambda: [{"id": "abc", "name": "notes.txt", "size": 2}])
|
||||
@@ -518,3 +519,294 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert (tmp_path / "notes.txt").read_bytes() == b"hi"
|
||||
|
||||
|
||||
# ── server-picked voice (the desk API's speak_as marker) ────────────────────
|
||||
|
||||
def _voice_turn(monkeypatch, ctrl, reply, said="talk like a pirate"):
|
||||
"""Run one full conversation turn whose reply is *reply*, returning the
|
||||
voice_id each tts.speak() call was given."""
|
||||
voices = []
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance",
|
||||
lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said)
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: reply)
|
||||
monkeypatch.setattr(
|
||||
controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None:
|
||||
voices.append(voice_id) or True,
|
||||
)
|
||||
ctrl._handle_conversation_turn()
|
||||
return voices
|
||||
|
||||
|
||||
def test_a_speak_as_reply_is_spoken_in_that_voice(monkeypatch, ctrl):
|
||||
voices = _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
|
||||
assert voices == ["VOICE1"]
|
||||
assert ctrl.current_voice() == "Terence"
|
||||
|
||||
|
||||
def test_the_picked_voice_sticks_for_later_replies(monkeypatch, ctrl):
|
||||
"""The server tags one reply and doesn't keep the id in its history, so
|
||||
it can't re-request the voice when you say "keep talking like that"."""
|
||||
monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", True)
|
||||
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
|
||||
voices = _voice_turn(monkeypatch, ctrl, Reply("Still me."), said="and now?")
|
||||
assert voices == ["VOICE1"]
|
||||
|
||||
|
||||
def test_voice_stickiness_can_be_turned_off(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", False)
|
||||
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
|
||||
voices = _voice_turn(monkeypatch, ctrl, Reply("Back to normal."), said="and now?")
|
||||
assert voices == [None]
|
||||
assert ctrl.current_voice() == ""
|
||||
|
||||
|
||||
def test_a_new_pick_replaces_the_old_one(monkeypatch, ctrl):
|
||||
changes = _capture(ctrl.voice_changed)
|
||||
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
|
||||
voices = _voice_turn(monkeypatch, ctrl, Reply("こんにちは。", "VOICE2", "Asahi"),
|
||||
said="say that in Japanese")
|
||||
assert voices == ["VOICE2"]
|
||||
assert changes == ["Terence", "Asahi"]
|
||||
|
||||
|
||||
def test_resetting_the_voice_goes_back_to_the_default(monkeypatch, ctrl):
|
||||
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
|
||||
changes = _capture(ctrl.voice_changed)
|
||||
|
||||
ctrl.reset_voice()
|
||||
|
||||
assert ctrl.current_voice() == ""
|
||||
assert changes == [""] # the tray's menu entry follows this signal
|
||||
voices = _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi")
|
||||
assert voices == [None]
|
||||
|
||||
|
||||
def test_an_unnamed_voice_still_reports_something_resettable(monkeypatch, ctrl):
|
||||
"""voice_name is optional server-side — falling back to the id keeps the
|
||||
tray entry from reading "now: " with nothing after it."""
|
||||
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1"))
|
||||
assert ctrl.current_voice() == "VOICE1"
|
||||
|
||||
|
||||
def test_petctl_voice_reset_returns_bolt_to_his_own_voice(monkeypatch, ctrl):
|
||||
"""The server can pick a voice but can't ask for the default back — it
|
||||
was never told what Bolt's own voice id is. This is how it asks."""
|
||||
ran = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
|
||||
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
|
||||
|
||||
output = ctrl._handle_command("petctl voice reset")
|
||||
|
||||
assert ran == [] # never reaches a shell, like every other petctl verb
|
||||
assert "Terence" in output # the server can't see the voice; tell it what changed
|
||||
assert ctrl.current_voice() == ""
|
||||
assert _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi") == [None]
|
||||
|
||||
|
||||
def test_petctl_voice_reset_says_so_when_there_was_nothing_to_reset(ctrl):
|
||||
assert "already" in ctrl._handle_command("petctl voice reset")
|
||||
|
||||
|
||||
# ── dialoguectl (multi-voice scenes) ────────────────────────────────────────
|
||||
|
||||
def _dialogue_command(*lines):
|
||||
import json
|
||||
return "dialoguectl " + json.dumps({"lines": list(lines)})
|
||||
|
||||
|
||||
def test_dialoguectl_never_reaches_the_shell(monkeypatch, ctrl):
|
||||
ran = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
|
||||
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
|
||||
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
|
||||
monkeypatch.setattr(controller_mod.tts, "play_pcm",
|
||||
lambda pcm, rate, should_stop=None: True)
|
||||
|
||||
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "[cheerfully] hi"}))
|
||||
|
||||
assert ran == []
|
||||
assert "[dialogue] played 1 line" in output
|
||||
|
||||
|
||||
def test_a_scene_shows_in_the_bubble_with_the_delivery_tags_stripped(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
|
||||
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
|
||||
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
|
||||
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
|
||||
said = _capture(ctrl.said)
|
||||
|
||||
ctrl._handle_command(_dialogue_command(
|
||||
{"voice": "self", "text": "[cheerfully] Hello there!"},
|
||||
{"voice": "narrator", "text": "[whispering] He is lying."},
|
||||
))
|
||||
|
||||
assert said == ["Hello there! He is lying."]
|
||||
assert ctrl.history.last().text == "Hello there! He is lying."
|
||||
|
||||
|
||||
def test_a_mid_turn_scene_returns_to_thinking_not_idle(monkeypatch, ctrl):
|
||||
"""The server is still waiting on the tool result, so the pet talks and
|
||||
goes back to waiting — dropping to IDLE would look like the turn ended."""
|
||||
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
|
||||
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
|
||||
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
|
||||
ctrl._state.transition(PetState.LISTENING)
|
||||
ctrl._state.transition(PetState.THINKING)
|
||||
states = _capture(ctrl.state_changed)
|
||||
|
||||
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
|
||||
|
||||
assert states == ["talking", "thinking"]
|
||||
assert ctrl._state.state == PetState.THINKING
|
||||
|
||||
|
||||
def test_the_scene_uses_a_voice_the_server_picked_with_speak_as(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
|
||||
seen = {}
|
||||
|
||||
def capture(inputs, model_id=None, stability=None):
|
||||
seen["inputs"] = inputs
|
||||
return np.zeros(4, dtype=np.int16), 24000
|
||||
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", capture)
|
||||
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
|
||||
ctrl._apply_voice(controller_mod.server_client.Reply("ok", "PICKEDvoice123456789", "Terence"))
|
||||
|
||||
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
|
||||
|
||||
assert seen["inputs"][0]["voice_id"] == "PICKEDvoice123456789"
|
||||
|
||||
|
||||
def test_a_synthesis_failure_is_reported_back_for_bolt_to_retry(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
|
||||
|
||||
def boom(inputs, model_id=None, stability=None):
|
||||
raise controller_mod.tts.TtsError("voice_id not found")
|
||||
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", boom)
|
||||
|
||||
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
|
||||
|
||||
assert "couldn't synthesize" in output and "voice_id not found" in output
|
||||
assert ctrl._state.state == PetState.IDLE # nothing left half-transitioned
|
||||
|
||||
|
||||
def test_a_bad_voice_name_comes_back_as_advice_not_an_exception(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
|
||||
output = ctrl._handle_command(_dialogue_command({"voice": "wizard", "text": "hi"}))
|
||||
assert "unknown voice" in output and "narrator" in output
|
||||
|
||||
|
||||
def test_dialogue_can_be_switched_off_on_this_device(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "DIALOGUE", False)
|
||||
called = []
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
|
||||
lambda *a, **k: called.append(1))
|
||||
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
|
||||
assert "disabled" in output and called == []
|
||||
|
||||
|
||||
def test_talking_over_a_scene_is_reported_up_the_relay(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
|
||||
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
|
||||
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
|
||||
monkeypatch.setattr(controller_mod.tts, "play_pcm",
|
||||
lambda pcm, rate, should_stop=None: False) # barge-in
|
||||
|
||||
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
|
||||
|
||||
assert "interrupted" in output
|
||||
|
||||
|
||||
# ── petctl self_restart ─────────────────────────────────────────────────────
|
||||
|
||||
def test_self_restart_arms_after_the_turn_rather_than_dying_mid_relay(monkeypatch, ctrl, tmp_path):
|
||||
"""Restarting inline would kill the HTTP tool relay before the result was
|
||||
posted, and the server would wait out its timeout on a turn that can never
|
||||
finish. So the command returns, the turn completes, *then* the pet dies."""
|
||||
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
|
||||
monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None)
|
||||
restarts = _capture(ctrl.restart_requested)
|
||||
|
||||
output = ctrl._handle_command("petctl self_restart check the new dialogue code")
|
||||
|
||||
assert "restarting as soon as this turn finishes" in output
|
||||
assert restarts == [] # nothing has happened yet
|
||||
|
||||
assert ctrl._maybe_self_restart() is True
|
||||
assert restarts and "check the new dialogue code" in restarts[0]
|
||||
|
||||
|
||||
def test_a_broken_edit_is_reported_instead_of_leaving_nothing_running(monkeypatch, ctrl, tmp_path):
|
||||
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
|
||||
|
||||
def boom(*args, **kwargs):
|
||||
raise controller_mod.self_restart.RestartError(
|
||||
"the current code does not import, so restarting would leave you with "
|
||||
"nothing running. Fix this first:\nSyntaxError: invalid syntax"
|
||||
)
|
||||
|
||||
monkeypatch.setattr(controller_mod.self_restart, "preflight", boom)
|
||||
restarts = _capture(ctrl.restart_requested)
|
||||
|
||||
output = ctrl._handle_command("petctl self_restart try the new code")
|
||||
|
||||
assert "SyntaxError" in output and "refused" in output
|
||||
assert ctrl._maybe_self_restart() is False
|
||||
assert restarts == []
|
||||
|
||||
|
||||
def test_a_second_restart_request_in_one_turn_is_a_no_op(monkeypatch, ctrl, tmp_path):
|
||||
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
|
||||
monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None)
|
||||
ctrl._handle_command("petctl self_restart first")
|
||||
assert "already armed" in ctrl._handle_command("petctl self_restart second")
|
||||
|
||||
|
||||
def test_self_restart_can_be_switched_off_on_this_device(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "SELF_RESTART", False)
|
||||
checked = []
|
||||
monkeypatch.setattr(controller_mod.self_restart, "preflight",
|
||||
lambda *a, **k: checked.append(1))
|
||||
assert "disabled" in ctrl._handle_command("petctl self_restart go")
|
||||
assert checked == []
|
||||
|
||||
|
||||
def test_coming_back_up_reports_to_the_server_and_speaks_the_reply(monkeypatch, ctrl, tmp_path):
|
||||
"""The half that makes it a loop: the new process tells Bolt it's back and
|
||||
why, and his answer is spoken like any other turn."""
|
||||
state = tmp_path / "ctx.json"
|
||||
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", state)
|
||||
controller_mod.self_restart.arm("check the walk cycle", version="0.2.3",
|
||||
path=state, now=1000.0)
|
||||
sent, spoken = [], []
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: sent.append(text) or Reply("Good, it's up."))
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None, voice_id=None:
|
||||
spoken.append(text) or True)
|
||||
|
||||
ctrl._report_self_restart()
|
||||
|
||||
assert "[pet self-restart]" in sent[0] and "check the walk cycle" in sent[0]
|
||||
assert spoken == ["Good, it's up."]
|
||||
# Consumed, so the next start doesn't announce the same restart again.
|
||||
assert controller_mod.self_restart.load(state) is None
|
||||
|
||||
|
||||
def test_an_ordinary_start_reports_nothing(monkeypatch, ctrl, tmp_path):
|
||||
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "none.json")
|
||||
called = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: called.append(text))
|
||||
|
||||
ctrl._report_self_restart()
|
||||
|
||||
assert called == []
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
"""`dialoguectl` — multi-voice scene parsing, voice resolution, API limits,
|
||||
and the request the ElevenLabs Text to Dialogue endpoint actually gets.
|
||||
|
||||
Pure logic plus one mocked HTTP call: no audio device, no network, no display.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import dialogue
|
||||
from bolt_pet.audio import tts
|
||||
|
||||
SELF_ID = "aaorr6ZHIL88gEexu7dC"
|
||||
NARRATOR_ID = "9BWtsMINqrJLrRacOk9x"
|
||||
VILLAIN_ID = "IKne3meq5aSn9XLyUdCD"
|
||||
VOICES = {"narrator": NARRATOR_ID, "villain": VILLAIN_ID}
|
||||
|
||||
|
||||
def _scene(*lines):
|
||||
return '{"lines": [' + ", ".join(lines) + "]}"
|
||||
|
||||
|
||||
# ── parsing ─────────────────────────────────────────────────────────────────
|
||||
|
||||
def test_non_dialogue_commands_are_left_alone():
|
||||
assert dialogue.parse("ls -la") is None
|
||||
assert dialogue.parse('filectl {"op": "list"}') is None
|
||||
assert dialogue.parse("") is None
|
||||
# "dialogues" must not be mistaken for the "dialogue" prefix
|
||||
assert dialogue.parse("dialogues --list") is None
|
||||
|
||||
|
||||
def test_a_scene_parses_into_lines():
|
||||
action = dialogue.parse(
|
||||
'dialoguectl ' + _scene(
|
||||
'{"voice": "self", "text": "[cheerfully] Hello, how are you?"}',
|
||||
'{"voice": "villain", "text": "[stuttering] I am... fine."}',
|
||||
)
|
||||
)
|
||||
assert action["action"] == "dialogue"
|
||||
assert [line["voice"] for line in action["lines"]] == ["self", "villain"]
|
||||
assert action["lines"][0]["text"].startswith("[cheerfully]")
|
||||
|
||||
|
||||
def test_the_elevenlabs_field_names_are_accepted_too():
|
||||
"""The model has read that API; copying its shape is the obvious thing to
|
||||
try, so 'inputs'/'voice_id' work as well as 'lines'/'voice'."""
|
||||
action = dialogue.parse(
|
||||
'dialoguectl {"inputs": [{"voice_id": "%s", "text": "hi"}]}' % NARRATOR_ID
|
||||
)
|
||||
assert action["lines"] == [{"voice": NARRATOR_ID, "text": "hi"}]
|
||||
|
||||
|
||||
def test_a_line_with_no_voice_defaults_to_the_pet_itself():
|
||||
action = dialogue.parse('dialoguectl {"lines": [{"text": "just me talking"}]}')
|
||||
assert action["lines"][0]["voice"] == "self"
|
||||
|
||||
|
||||
def test_truncated_json_explains_the_one_line_rule():
|
||||
"""The real failure mode: the server's command extractor stops at the
|
||||
first newline, so a multi-line payload arrives cut in half. The error has
|
||||
to name the cause, since Bolt is the one who has to fix it."""
|
||||
with pytest.raises(dialogue.DialogueError, match="one line"):
|
||||
dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"')
|
||||
|
||||
|
||||
def test_an_empty_or_shapeless_payload_is_rejected():
|
||||
with pytest.raises(dialogue.DialogueError, match="needs a JSON argument"):
|
||||
dialogue.parse("dialoguectl")
|
||||
with pytest.raises(dialogue.DialogueError, match="non-empty"):
|
||||
dialogue.parse('dialoguectl {"lines": []}')
|
||||
with pytest.raises(dialogue.DialogueError, match="no text"):
|
||||
dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": " "}]}')
|
||||
|
||||
|
||||
def test_optional_model_and_stability_ride_along():
|
||||
action = dialogue.parse(
|
||||
'dialoguectl {"model_id": "eleven_v3", "stability": 0.8, '
|
||||
'"lines": [{"text": "hi"}]}'
|
||||
)
|
||||
assert action["model"] == "eleven_v3"
|
||||
assert action["stability"] == 0.8
|
||||
|
||||
|
||||
# ── voice resolution ────────────────────────────────────────────────────────
|
||||
|
||||
def test_named_voices_resolve_from_the_configured_cast():
|
||||
action = dialogue.parse('dialoguectl ' + _scene(
|
||||
'{"voice": "narrator", "text": "Once upon a time."}',
|
||||
'{"voice": "villain", "text": "Not this again."}',
|
||||
))
|
||||
inputs = dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID)
|
||||
assert [entry["voice_id"] for entry in inputs] == [NARRATOR_ID, VILLAIN_ID]
|
||||
|
||||
|
||||
def test_self_tracks_the_voice_the_pet_is_currently_using():
|
||||
"""A scene featuring Bolt should sound like whoever Bolt currently is —
|
||||
including a voice the server picked mid-conversation with speak_as."""
|
||||
action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}')
|
||||
picked = "VOICEfromSPEAKas1234"
|
||||
assert dialogue.resolve(action, self_voice=picked)[0]["voice_id"] == picked
|
||||
|
||||
|
||||
def test_a_raw_voice_id_passes_straight_through():
|
||||
action = dialogue.parse('dialoguectl {"lines": [{"voice": "%s", "text": "hi"}]}' % NARRATOR_ID)
|
||||
assert dialogue.resolve(action, self_voice=SELF_ID)[0]["voice_id"] == NARRATOR_ID
|
||||
|
||||
|
||||
def test_an_unknown_name_lists_what_is_available():
|
||||
action = dialogue.parse('dialoguectl {"lines": [{"voice": "wizard", "text": "hi"}]}')
|
||||
with pytest.raises(dialogue.DialogueError) as excinfo:
|
||||
dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID)
|
||||
message = str(excinfo.value)
|
||||
assert "wizard" in message and "narrator" in message and "villain" in message
|
||||
|
||||
|
||||
def test_self_without_a_configured_voice_says_so():
|
||||
action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}')
|
||||
with pytest.raises(dialogue.DialogueError, match="ELEVENLABS_VOICE_ID"):
|
||||
dialogue.resolve(action, self_voice="")
|
||||
|
||||
|
||||
def test_the_voice_map_parser_skips_typos_instead_of_dying():
|
||||
voices = dialogue.parse_voice_map(f"narrator:{NARRATOR_ID}, broken-entry, villain:{VILLAIN_ID}")
|
||||
assert voices == {"narrator": NARRATOR_ID, "villain": VILLAIN_ID}
|
||||
assert dialogue.parse_voice_map("") == {}
|
||||
|
||||
|
||||
# ── API limits, enforced before the request goes out ────────────────────────
|
||||
|
||||
def test_too_many_distinct_voices_is_refused_locally():
|
||||
inputs = [{"text": "hi", "voice_id": f"voice{index:015d}"} for index in range(11)]
|
||||
with pytest.raises(dialogue.DialogueError, match="limit is 10"):
|
||||
dialogue.check_limits(inputs)
|
||||
|
||||
|
||||
def test_an_over_long_scene_is_refused_with_advice():
|
||||
inputs = [{"text": "x" * 1100, "voice_id": SELF_ID} for _ in range(2)]
|
||||
with pytest.raises(dialogue.DialogueError) as excinfo:
|
||||
dialogue.check_limits(inputs)
|
||||
assert "Split it" in str(excinfo.value) # actionable, since Bolt reads this
|
||||
|
||||
|
||||
# ── display / reporting ─────────────────────────────────────────────────────
|
||||
|
||||
def test_delivery_tags_are_stripped_from_what_the_bubble_shows():
|
||||
action = dialogue.parse('dialoguectl ' + _scene(
|
||||
'{"voice": "self", "text": "[cheerfully] Hello there!"}',
|
||||
'{"voice": "narrator", "text": "[whispering] He is lying."}',
|
||||
))
|
||||
assert dialogue.spoken_text(action) == "Hello there! He is lying."
|
||||
|
||||
|
||||
def test_the_relay_report_names_the_cast():
|
||||
action = dialogue.parse('dialoguectl ' + _scene(
|
||||
'{"voice": "self", "text": "one"}', '{"voice": "narrator", "text": "two"}',
|
||||
))
|
||||
assert dialogue.describe(action) == "[dialogue] played 2 lines in 2 voices: narrator, self"
|
||||
|
||||
|
||||
# ── the HTTP request ────────────────────────────────────────────────────────
|
||||
|
||||
def _pcm_response(samples=(1, 2, 3, 4)):
|
||||
response = MagicMock()
|
||||
response.content = np.array(samples, dtype=np.int16).tobytes()
|
||||
response.raise_for_status = MagicMock()
|
||||
return response
|
||||
|
||||
|
||||
def test_the_request_matches_the_text_to_dialogue_api(monkeypatch):
|
||||
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
|
||||
monkeypatch.setattr(tts.config, "TTS_SAMPLE_RATE", 24000)
|
||||
monkeypatch.setattr(tts.config, "DIALOGUE_MODEL_ID", "eleven_v3")
|
||||
inputs = [
|
||||
{"text": "[cheerfully] Hello", "voice_id": NARRATOR_ID},
|
||||
{"text": "[stuttering] H-hi", "voice_id": VILLAIN_ID},
|
||||
]
|
||||
with patch.object(tts.requests, "post", return_value=_pcm_response()) as post:
|
||||
pcm, rate = tts.synthesize_dialogue(inputs)
|
||||
|
||||
assert rate == 24000 and pcm.tolist() == [1, 2, 3, 4]
|
||||
args, kwargs = post.call_args
|
||||
assert args[0] == "https://api.elevenlabs.io/v1/text-to-dialogue"
|
||||
assert kwargs["params"] == {"output_format": "pcm_24000"}
|
||||
assert kwargs["headers"] == {"xi-api-key": "test-key"}
|
||||
assert kwargs["json"]["inputs"] == inputs
|
||||
assert kwargs["json"]["model_id"] == "eleven_v3"
|
||||
assert "settings" not in kwargs["json"] # omitted unless asked for
|
||||
|
||||
|
||||
def test_stability_is_only_sent_when_given(monkeypatch):
|
||||
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
|
||||
with patch.object(tts.requests, "post", return_value=_pcm_response()) as post:
|
||||
tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}], stability=0.3)
|
||||
assert post.call_args.kwargs["json"]["settings"] == {"stability": 0.3}
|
||||
|
||||
|
||||
def test_a_rejected_request_surfaces_what_the_api_said(monkeypatch):
|
||||
"""The API explains refusals in the body; Bolt reads this through the tool
|
||||
relay, so it has to reach him rather than being flattened to '422'."""
|
||||
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
|
||||
failure = MagicMock()
|
||||
failure.text = '{"detail": "voice_id not found"}'
|
||||
error = Exception("422 Client Error")
|
||||
error.response = failure
|
||||
response = MagicMock()
|
||||
response.raise_for_status = MagicMock(side_effect=error)
|
||||
|
||||
with patch.object(tts.requests, "post", return_value=response):
|
||||
with pytest.raises(tts.TtsError, match="voice_id not found"):
|
||||
tts.synthesize_dialogue([{"text": "hi", "voice_id": "nope"}])
|
||||
|
||||
|
||||
def test_no_api_key_fails_before_the_request(monkeypatch):
|
||||
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "")
|
||||
with patch.object(tts.requests, "post") as post:
|
||||
with pytest.raises(tts.TtsError):
|
||||
tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}])
|
||||
post.assert_not_called()
|
||||
@@ -69,3 +69,17 @@ def test_describe_is_reported_back_to_the_server():
|
||||
assert "top-left" in pet_actions.describe({"action": "move", "anchor": "top-left"})
|
||||
assert "wave" in pet_actions.describe({"action": "emote", "emote": "wave"})
|
||||
assert pet_actions.describe({"action": "help"}) == pet_actions.HELP
|
||||
|
||||
|
||||
def test_voice_reset_parses_with_or_without_the_word_reset():
|
||||
assert pet_actions.parse("petctl voice reset") == {"action": "voice", "voice": "default"}
|
||||
assert pet_actions.parse("petctl voice default") == {"action": "voice", "voice": "default"}
|
||||
assert pet_actions.parse("petctl voice") == {"action": "voice", "voice": "default"}
|
||||
|
||||
|
||||
def test_petctl_cannot_be_used_to_pick_a_voice():
|
||||
"""Choosing a voice is the server's job (speak_as) — it has the voice
|
||||
library. petctl only ever undoes one, so an attempt to set a voice here
|
||||
is pointed back at the marker that works."""
|
||||
with pytest.raises(pet_actions.ActionError, match="speak_as"):
|
||||
pet_actions.parse("petctl voice Terence")
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
"""`petctl self_restart` — the pet restarting itself and remembering why.
|
||||
|
||||
Everything here runs against a temp context file and a fake subprocess runner,
|
||||
so the tests exercise the arming/preflight/report logic without any process
|
||||
actually dying.
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import pet_actions, self_restart
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def state(tmp_path):
|
||||
return tmp_path / "restart_context.json"
|
||||
|
||||
|
||||
def _ok_run(*args, **kwargs):
|
||||
return SimpleNamespace(returncode=0, stdout="", stderr="")
|
||||
|
||||
|
||||
def _broken_run(*args, **kwargs):
|
||||
return SimpleNamespace(
|
||||
returncode=1, stdout="",
|
||||
stderr=' File "bolt_pet/controller.py", line 42\n def _speak(\nSyntaxError: invalid syntax',
|
||||
)
|
||||
|
||||
|
||||
# ── parsing ─────────────────────────────────────────────────────────────────
|
||||
|
||||
def test_self_restart_parses_with_a_free_text_reason():
|
||||
action = pet_actions.parse("petctl self_restart check the new walk cycle loads")
|
||||
assert action == {
|
||||
"action": "self_restart", "reason": "check the new walk cycle loads",
|
||||
}
|
||||
|
||||
|
||||
def test_self_restart_needs_no_reason_and_accepts_aliases():
|
||||
assert pet_actions.parse("petctl self_restart")["reason"] == ""
|
||||
assert pet_actions.parse("petctl restart")["action"] == "self_restart"
|
||||
assert pet_actions.parse("petctl reboot")["action"] == "self_restart"
|
||||
|
||||
|
||||
def test_self_restart_is_listed_in_the_help():
|
||||
assert "self_restart" in pet_actions.HELP
|
||||
|
||||
|
||||
# ── preflight ───────────────────────────────────────────────────────────────
|
||||
|
||||
def test_preflight_passes_when_the_code_imports():
|
||||
self_restart.preflight(run=_ok_run) # no exception
|
||||
|
||||
|
||||
def test_preflight_hands_back_the_traceback_instead_of_dying(state):
|
||||
"""The whole point: a syntax error Bolt just introduced comes back as
|
||||
something he can read and fix, in the same turn, with the pet still up."""
|
||||
with pytest.raises(self_restart.RestartError) as excinfo:
|
||||
self_restart.preflight(run=_broken_run)
|
||||
message = str(excinfo.value)
|
||||
assert "does not import" in message
|
||||
assert "SyntaxError" in message and "controller.py" in message
|
||||
|
||||
|
||||
def test_preflight_runs_the_import_in_a_subprocess_not_here():
|
||||
"""This process holds the *old* modules, so an in-process import would
|
||||
pass on a file that no longer parses."""
|
||||
seen = {}
|
||||
|
||||
def capture(cmd, **kwargs):
|
||||
seen["cmd"], seen["kwargs"] = cmd, kwargs
|
||||
return SimpleNamespace(returncode=0, stdout="", stderr="")
|
||||
|
||||
self_restart.preflight(run=capture)
|
||||
assert seen["cmd"][0] == sys.executable
|
||||
assert "import bolt_pet" in seen["cmd"][2]
|
||||
assert seen["kwargs"]["env"]["QT_QPA_PLATFORM"] == "offscreen" # imports need no display
|
||||
|
||||
|
||||
def test_a_subprocess_that_cannot_even_run_is_reported(monkeypatch):
|
||||
def explode(*args, **kwargs):
|
||||
raise OSError("no python here")
|
||||
|
||||
with pytest.raises(self_restart.RestartError, match="couldn't run the preflight"):
|
||||
self_restart.preflight(run=explode)
|
||||
|
||||
|
||||
# ── context across the restart ──────────────────────────────────────────────
|
||||
|
||||
def test_arming_persists_the_reason_for_the_next_process(state):
|
||||
self_restart.arm("check the sprite frames load", version="0.2.3",
|
||||
session="pet-desktop", recent=["you: reload the sprites"],
|
||||
path=state, now=1000.0)
|
||||
revived = self_restart.load(state)
|
||||
assert revived.reason == "check the sprite frames load"
|
||||
assert revived.version == "0.2.3"
|
||||
assert revived.recent == ["you: reload the sprites"]
|
||||
assert revived.restarts == [1000.0]
|
||||
|
||||
|
||||
def test_no_context_means_a_normal_start(state):
|
||||
assert self_restart.load(state) is None
|
||||
|
||||
|
||||
def test_a_corrupt_context_file_is_ignored_not_fatal(state):
|
||||
state.write_text("{not json at all", encoding="utf-8")
|
||||
assert self_restart.load(state) is None
|
||||
|
||||
|
||||
def test_clearing_the_context_stops_it_being_re_announced(state):
|
||||
self_restart.arm("once", path=state, now=1000.0)
|
||||
self_restart.clear(state)
|
||||
assert self_restart.load(state) is None
|
||||
self_restart.clear(state) # clearing twice is not an error
|
||||
|
||||
|
||||
def test_the_report_says_what_happened_and_what_to_check(state):
|
||||
context = self_restart.arm(
|
||||
"verify the dialogue command works", verify="verify the dialogue command works",
|
||||
version="0.2.3", recent=["you: try a scene"], path=state, now=1000.0,
|
||||
)
|
||||
text = self_restart.report(context, version="0.2.4", now=1004.5)
|
||||
assert "I restarted myself" in text
|
||||
assert "verify the dialogue command works" in text
|
||||
assert "4.5s" in text
|
||||
assert "0.2.4" in text and "was 0.2.3" in text
|
||||
assert "you: try a scene" in text
|
||||
|
||||
|
||||
# ── loop guard ──────────────────────────────────────────────────────────────
|
||||
|
||||
def test_restart_history_accumulates_across_restarts(state):
|
||||
self_restart.arm("one", path=state, now=1000.0)
|
||||
self_restart.arm("two", path=state, now=1100.0)
|
||||
assert self_restart.load(state).restarts == [1000.0, 1100.0]
|
||||
|
||||
|
||||
def test_too_many_restarts_in_the_window_is_refused(state):
|
||||
now = 1000.0
|
||||
for index in range(self_restart.MAX_RESTARTS):
|
||||
self_restart.arm(f"attempt {index}", path=state, now=now + index)
|
||||
with pytest.raises(self_restart.RestartError, match="looping"):
|
||||
self_restart.check_loop_guard(self_restart.load(state), now=now + 10)
|
||||
|
||||
|
||||
def test_old_restarts_fall_out_of_the_window(state):
|
||||
now = 1000.0
|
||||
for index in range(self_restart.MAX_RESTARTS):
|
||||
self_restart.arm(f"attempt {index}", path=state, now=now + index)
|
||||
later = now + self_restart.WINDOW_SECONDS + 60
|
||||
self_restart.check_loop_guard(self_restart.load(state), now=later) # no exception
|
||||
assert self_restart.recent_restarts(self_restart.load(state), now=later) == []
|
||||
@@ -27,7 +27,8 @@ def test_converse_returns_reply_directly():
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({"type": "reply", "text": "hello there"})
|
||||
result = server_client.converse("hi")
|
||||
assert result == "hello there"
|
||||
assert result.text == "hello there"
|
||||
assert result.voice_id == "" # no speak_as on this reply
|
||||
post.assert_called_once()
|
||||
args, kwargs = post.call_args
|
||||
assert args[0] == "http://test-server:5002/desk/converse"
|
||||
@@ -43,7 +44,7 @@ def test_converse_relays_a_command_then_returns_reply():
|
||||
with patch.object(server_client.requests, "post", side_effect=responses) as post:
|
||||
on_command = MagicMock(return_value="[exit 0]\nhi")
|
||||
result = server_client.converse("run echo hi", on_command=on_command)
|
||||
assert result == "done"
|
||||
assert result.text == "done"
|
||||
on_command.assert_called_once_with("echo hi")
|
||||
# second call was to /desk/tool_result with the command's output
|
||||
second_call = post.call_args_list[1]
|
||||
@@ -53,6 +54,19 @@ def test_converse_relays_a_command_then_returns_reply():
|
||||
}
|
||||
|
||||
|
||||
def test_converse_carries_a_speak_as_voice_back_with_the_reply():
|
||||
"""The server tags a reply with the voice it picked (speak_as); this
|
||||
client is what actually speaks in it, so the id has to survive the
|
||||
return trip rather than being dropped with the rest of the payload."""
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({
|
||||
"type": "reply", "text": "Ahoy there.",
|
||||
"voice_id": "hnhGxwvHP8fc469w51rM", "voice_name": "Terence",
|
||||
})
|
||||
result = server_client.converse("talk like a pirate")
|
||||
assert result == server_client.Reply("Ahoy there.", "hnhGxwvHP8fc469w51rM", "Terence")
|
||||
|
||||
|
||||
def test_converse_raises_server_error_on_error_payload():
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({"type": "error", "error": "unauthorized"})
|
||||
|
||||
@@ -8,7 +8,8 @@ import numpy as np
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet.audio.tts import chunks_to_int16
|
||||
from bolt_pet import config as tts_config
|
||||
from bolt_pet.audio.tts import chunks_to_int16, model_for, voice_for
|
||||
from bolt_pet.audio.wake_word import NearMissLog
|
||||
|
||||
|
||||
@@ -80,3 +81,27 @@ def test_clear_resets_peak_and_entries():
|
||||
log.observe(0.45, threshold=0.5, timestamp=1.0)
|
||||
log.clear()
|
||||
assert log.entries() == [] and log.peak == 0.0
|
||||
|
||||
|
||||
# ── voice / model selection (server speak_as) ───────────────────────────────
|
||||
|
||||
def test_the_override_voice_wins_over_the_configured_one(monkeypatch):
|
||||
monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "DEFAULT")
|
||||
assert voice_for("VOICE1") == "VOICE1"
|
||||
assert voice_for("") == "DEFAULT"
|
||||
assert voice_for(None) == "DEFAULT"
|
||||
|
||||
|
||||
def test_english_replies_in_the_default_voice_use_the_default_model(monkeypatch):
|
||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
|
||||
assert model_for("all good here", None) == "eleven_flash_v2"
|
||||
|
||||
|
||||
def test_a_picked_voice_or_non_english_text_uses_the_multilingual_model(monkeypatch):
|
||||
# eleven_flash_v2 is English-only: it would read either of these as
|
||||
# mangled phonetic English rather than failing outright.
|
||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
||||
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
|
||||
assert model_for("all good here", "VOICE1") == "eleven_flash_v2_5"
|
||||
assert model_for("こんにちは", None) == "eleven_flash_v2_5"
|
||||
|
||||
Reference in New Issue
Block a user