11 Commits

Author SHA1 Message Date
themajesticmagician 3ee67cb4d6 feat: Enhance local command handling and introduce local intents
- Refactor `run_local_command` to manage subprocesses more effectively, ensuring child processes are terminated on timeout.
- Introduce `_terminate` function to handle process group termination and capture output.
- Implement `_command_output` to format command results with a character limit.
- Add local intent recognition in `intents.py` to handle commands like "stop", "go to sleep", and "come here" without server interaction.
- Normalize user input to match local intents while stripping filler words.
- Update tests to cover new local intent functionality and ensure proper command handling.
- Enhance speech processing to handle abbreviations and improve spoken output clarity.
2026-08-05 18:31:02 -06:00
themajesticmagician 8d4751d80f Update CLAUDE.md with release tagging instructions and enhance is_question logic to detect question marks anywhere in the text 2026-08-05 17:53:32 -06:00
themajesticmagician c4e805defd Add relay_json module, update dialogue and file_ops, update local settings 2026-07-31 01:27:20 -06:00
themajesticmagician 96afc351ac Add text-to-dialogue, self-restart capability, and misc updates 2026-07-30 20:50:41 -06:00
themajesticmagician 5b49670983 v0.2.3
Multi-monitor jumps (`petctl jump`/`monitors`), pull-only screen OCR
(`petctl read`), generated sprite art with a distance-stepped walk cycle,
plus the filectl file ops and server file delivery merged back in.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-07-28 16:20:46 -06:00
themajesticmagician 311fe8b709 Merge remote-tracking branch 'origin/main' into screens
# Conflicts:
#	.claude/settings.local.json
#	CLAUDE.md
#	bolt_pet/controller.py
2026-07-28 16:19:45 -06:00
themajesticmagician b121bbba17 Multi-monitor jumps, screen OCR, and generated sprite art
petctl gains screen verbs: `jump` (1-based number, name, next/prev/
primary/other, or a direction resolved from real geometry), `monitors`,
and `read` for OCR of a monitor's contents.

- monitors.py: pure layout model + jump-target resolution. The monitor
  list is published by PetWindow from QGuiApplication.screens() over a
  queued signal, so the controller and window agree on what "monitor 2"
  means; xrandr and Qt order screens differently on the same machine.
- screen_text.py: pull-only OCR (mss capture + Tesseract/RapidOCR).
  Nothing captures unless the server asks, and the text rides back up
  the tool-result relay so Bolt can read a screen mid-turn. Both deps
  optional, soft-failing with a reason. SCREEN_TEXT=false removes it.
- Query verbs are answered in controller._handle_command rather than
  pet_actions.describe(), because their output is the point.
- scripts/generate_bolt_sprites.py draws every frame; walk/ is a
  side-view cycle stepped by distance travelled, not by the animation
  timer, so the planted paw tracks the window exactly. sprite.py loads
  it via EXTRA_ANIMATIONS keyed by name, with has() so callers can
  decline a placeholder blob.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-07-28 16:17:40 -06:00
themajesticmagician ccb3aeb7ae v0.2.2 2026-07-26 19:10:14 -06:00
themajesticmagician c16fada8d8 Update 2026-07-26 18:49:51 -06:00
themajesticmagician e9d92b0ba2 ... 2026-07-23 07:55:43 -06:00
themajesticmagician 843f52c507 Wake-word barge-in, Gitea auto-updater, hard_reset fix 2026-07-23 07:30:08 -06:00
84 changed files with 8383 additions and 189 deletions
+67 -1
View File
@@ -26,7 +26,73 @@
"Bash(.venv/bin/python *)",
"Bash(python *)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python -m pytest tests/test_controller_features.py -q -p no:cacheprovider)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python -c ' *)"
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python -c ' *)",
"Bash(git push *)",
"Bash(git remote *)",
"Bash(grep -v '^$')",
"Bash(.venv/bin/pip install *)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python *)",
"Bash(python3 *)",
"Bash(.venv/bin/pytest tests/test_monitors.py -q)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_pet_window_features.py -q)",
"Bash(git fetch *)",
"Bash(git switch *)",
"Bash(git add *)",
"Bash(git merge *)",
"Bash(echo \"=== EXIT: $? ===\")",
"Bash(git commit *)",
"Bash(QT_QPA_PLATFORM=offscreen /root/Documents/bolt-pet/.venv/bin/pytest tests/ -q)",
"Bash(.venv/bin/pytest tests/test_desk_api.py tests/test_desk_files.py tests/test_desk_voice.py tests/test_desk_status.py tests/test_desk_keys.py tests/test_desk_guild_action.py tests/test_desk_admin.py tests/test_desk_billing_auth.py -q)",
"Bash(docker inspect *)",
"Bash(python3 -m json.tool)",
"Bash(docker restart bolt *)",
"Bash(curl -s -m 5 \"http://localhost:5002/desk/health\")",
"Bash(python3 -c ' *)",
"Bash(QT_QPA_PLATFORM=offscreen /root/Documents/bolt-pet/.venv/bin/pytest /home/themajesticmagician/Documents/Bolt-Pet/tests/ -q)",
"Bash(docker exec bolt *)",
"Bash(QT_QPA_PLATFORM=offscreen /root/Documents/bolt-pet/.venv/bin/pytest /home/themajesticmagician/Documents/Bolt-Pet/tests/test_file_ops.py -q)",
"Bash(grep -rn *)",
"Bash(git ls-tree *)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_controller_features.py -q)",
"Read(//home/maji/Documents/tmn-api/**)",
"Bash(timeout 300 .venv/bin/python -m pytest tests/test_desk_voice.py -q)",
"Bash(echo \"exit=$?\")",
"Bash(timeout 300 /home/maji/Documents/tmn-api/.venv/bin/python -m pytest /home/maji/Documents/tmn-api/tests/test_desk_voice.py -q -p no:cacheprovider --rootdir=/home/maji/Documents/tmn-api)",
"Bash(echo \"EXIT=$?\")",
"Bash(/home/maji/Documents/tmn-api/.venv/bin/python -c \"import ast,pathlib; ast.parse\\(pathlib.Path\\('/home/maji/Documents/tmn-api/ai/desk_api.py'\\).read_text\\(\\)\\); print\\('desk_api.py parses OK'\\)\")",
"Bash(ps -eo pid,etime,cmd)",
"Bash(systemctl --user list-units --type=service)",
"Read(//run/user/1000/gvfs/sftp:host=192.168.2.231,user=root/Main/Docker-Compose/TMN-API/tmn-api/ai/**)",
"Bash(findmnt -T /home/maji/Documents/tmn-api -o TARGET,SOURCE,FSTYPE)",
"Bash(echo \"rc=$?\")",
"Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py -q -p no:cacheprovider)",
"Bash(timeout 600 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py -q -p no:cacheprovider)",
"Bash(awk 'NR>=1150 && NR<=1310 && \\(/return / || /def /\\)' ai/agents/default.py)",
"Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python -c ' *)",
"Bash(timeout 900 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py -q -p no:cacheprovider)",
"Bash(timeout 300 /home/maji/Documents/Bolt-Pet/.venv/bin/python -m pytest tests/test_proactive.py::test_send_trims_swallowed_tool_lines -q -p no:cacheprovider)",
"Bash(/home/maji/Documents/Bolt-Pet/.venv/bin/python *)",
"Bash(./tmnvenv/bin/pip install *)",
"Bash(./tmnvenv/bin/python -c \"import pytest,dotenv,yaml; print\\('scratch venv ready'\\)\")",
"Bash(/tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/pip install *)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_default_agent.py -q -p no:cacheprovider)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py -q -p no:cacheprovider)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_emotional_memory.py tests/test_inner_monologue.py tests/test_temporal_core.py tests/test_temporal_episodes.py tests/test_temporal_stream.py tests/test_temporal_facade.py tests/test_proactive.py tests/test_desk_api.py tests/test_main_helpers.py -q -p no:cacheprovider)",
"Bash(timeout 1800 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests -q -p no:cacheprovider --ignore=tests/test_tool_markers.py --ignore=tests/test_tts_sanitization.py --ignore=tests/test_speaker_matching.py --ignore=tests/test_recent_speaker_fallback.py --ignore=tests/test_assistant_cli_call_proxy.py --ignore=tests/test_assistant_cli_permissions.py --ignore=tests/test_memory_store.py --ignore=tests/test_default_agent.py --ignore=tests/test_billing_web.py)",
"WebFetch(domain:elevenlabs.io)",
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_dialogue.py -q)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_system_index.py tests/test_initiative.py tests/test_self_experiments.py -q -p no:cacheprovider)",
"Bash($V *)",
"Bash(dig +short themajesticnetwork.com)",
"Bash(dig +short api.themajesticnetwork.com)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_site.py -q -p no:cacheprovider)",
"Bash(curl -s -o /dev/null -w 'HTTP %{http_code} bytes=%{size_download}\\\\n' -m 15 -H 'X-Forwarded-For: 1.2.3.4' -H 'X-Real-IP: 1.2.3.4' -A 'Mozilla/5.0 \\(X11; Linux x86_64\\) Firefox/152.0' https://themajesticnetwork.com/?claude-probe-__TRACKED_VAR__)",
"Bash(QT_QPA_PLATFORM=offscreen timeout 300 .venv/bin/pytest tests/ -q)",
"Bash(QT_QPA_PLATFORM=offscreen timeout 120 .venv/bin/pytest tests/test_intents.py -q)",
"Bash(QT_QPA_PLATFORM=offscreen timeout 120 .venv/bin/pytest tests/test_speech_text.py -q)"
]
}
}
+128 -5
View File
@@ -31,6 +31,36 @@ ELEVENLABS_VOICE_ID=
#ELEVENLABS_MODEL_ID=eleven_flash_v2
#TTS_SAMPLE_RATE=24000
# Ask Bolt to use a different voice (or another language) and the server
# picks one from the ElevenLabs voice library and tags the reply with it.
# It only tags one reply, and it can't remember the id afterwards — so the
# pet keeps using that voice until a new one is picked or you choose "Use
# default voice" in the tray. VOICE_STICKY=false makes each pick last for
# exactly the one reply it came with instead.
#VOICE_STICKY=true
# ── Multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
# Lets Bolt play a short scene in several voices with delivery tags the v3
# model acts on ("[cheerfully] Hello", "[whispering] He is lying"), driven by
# the server through a relayed `dialoguectl` command. Name the cast here —
# "self" always means whatever voice the pet is currently using.
#DIALOGUE=true
#DIALOGUE_MODEL_ID=eleven_v3
#DIALOGUE_VOICES=narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD
# ── Self-restart (optional) ─────────────────────────────────────────────────
# `petctl self_restart <why>` lets Bolt reload the pet after editing its own
# code, so he can check the change live. The code is import-checked first, the
# restart waits for the current turn to finish, and the reason is carried
# across so the new process reports back. The guard refuses more than
# SELF_RESTART_MAX restarts within SELF_RESTART_WINDOW_SECONDS.
#SELF_RESTART=true
#SELF_RESTART_MAX=5
#SELF_RESTART_WINDOW_SECONDS=900
# Used instead of ELEVENLABS_MODEL_ID whenever the server picked the voice
# or the reply has non-ASCII in it — the flash_v2 default is English-only.
#ELEVENLABS_MULTILINGUAL_MODEL_ID=eleven_flash_v2_5
# ── Audio devices (optional — leave blank for the system default) ──────────
#MIC_DEVICE=
#SPEAKER_DEVICE=
@@ -40,6 +70,26 @@ ELEVENLABS_VOICE_ID=
#VAD_SILENCE_END_SEC=1.2
#VAD_MAX_UTTERANCE_SECONDS=15
#VAD_MIN_UTTERANCE_SECONDS=0.4
#VAD_GRACE_SECONDS=4 # how long to wait for you to start talking
# ── Local intents (optional) ────────────────────────────────────────────────
# A short, closed list of things the pet answers itself, with no server round
# trip: "stop", "be quiet", "come here", "go away", "go to sleep", "wake up",
# "say that again", "sit"/"stay", "go for a walk", "use your normal voice".
# Matched whole and exact, and never while you're answering a question Bolt
# asked, so a real request ("stop the docker container") still goes to him.
# Set to false to route absolutely everything through the server.
#LOCAL_INTENTS=true
# ── Follow-up listening (optional) ──────────────────────────────────────────
# When a reply asks you something, the pet keeps listening for your answer
# instead of dropping back to idle and making you say the wake word again.
# FOLLOW_UP_MAX_TURNS caps how many question-and-answer rounds can chain
# without you re-triggering it (0 = no cap) — a stop on runaway loops if the
# server ends every reply with "?" and the mic keeps feeding it noise.
#FOLLOW_UP_LISTEN=true
#FOLLOW_UP_MAX_TURNS=3
#FOLLOW_UP_GRACE_SECONDS=7 # longer than VAD_GRACE_SECONDS: you were asked something
# ── Pet window (optional) ───────────────────────────────────────────────────
#PET_SIZE=160
@@ -63,12 +113,36 @@ ELEVENLABS_VOICE_ID=
#PET_EDGE_SNAP=true
#PET_SNAP_MARGIN=48
# ── Barge-in (optional) — talk over the pet to cut it off ───────────────────
# Threshold defaults to 4x VAD_RMS_THRESHOLD because the mic also hears the
# pet's own voice out of the speakers. Raise it if playback self-interrupts.
# ── Barge-in (optional) — interrupt the pet mid-sentence ────────────────────
# BARGE_IN_MODE decides what counts as an interruption:
# wake — only the wake word cuts playback (default). Background noise,
# coughs and the TV can't stop it mid-sentence.
# energy — any sustained noise above BARGE_IN_RMS_THRESHOLD does. Faster to
# trigger, but interrupts on anything loud. That threshold defaults
# to 4x VAD_RMS_THRESHOLD because the mic also hears the pet's own
# voice out of the speakers; raise it if playback self-interrupts.
#BARGE_IN=true
#BARGE_IN_RMS_THRESHOLD=1200
#BARGE_IN_FRAMES=4
#BARGE_IN_MODE=wake
#BARGE_IN_RMS_THRESHOLD=1200 # energy mode only
#BARGE_IN_FRAMES=4 # energy mode only
# Wake mode only. Blank tracks the live WAKE_WORD_THRESHOLD (tray tuner);
# set a number to make interrupting harder than waking the pet from idle,
# e.g. if Bolt's own voice occasionally trips the model.
#BARGE_IN_WAKE_THRESHOLD=0.6
# ── Auto-update (optional) ──────────────────────────────────────────────────
# Watches the Gitea releases page for a tag newer than bolt_pet.__version__,
# then `git checkout`s it and restarts — only ever between turns, never
# mid-conversation. Requires the install to be a git clone; a working tree
# with local changes is skipped (never stashed), and any failure after the
# checkout rolls back to the ref that was live before.
#AUTO_UPDATE=true
#UPDATE_REPO_API=https://git.themajesticnetwork.com/api/v1/repos/TheMajesticNetwork/Bolt-Pet
#UPDATE_CHECK_INTERVAL_SECONDS=3600
#UPDATE_GIT_REMOTE=origin
#UPDATE_INSTALL_DEPS=true
# Only needed if the repo is private (a Gitea token with read:repository).
#UPDATE_TOKEN=
# ── Streaming TTS (optional) — starts talking on the first chunk ────────────
#TTS_STREAMING=true
@@ -78,6 +152,30 @@ ELEVENLABS_VOICE_ID=
# error?" has a referent. Text only — no screenshots leave the machine.
#SCREEN_CONTEXT=true
# ── Monitors (optional) ─────────────────────────────────────────────────────
# Tacks a one-line summary of your screen layout onto each utterance (how
# many, their sizes, which one the pet is standing on) so Bolt can decide to
# `petctl jump 2` without asking what you've got plugged in. Costs nothing —
# the list comes from the UI, nothing is probed per turn.
#MONITOR_CONTEXT=true
# ── Screen text / OCR (optional) ────────────────────────────────────────────
# Lets Bolt read what's actually on a monitor with `petctl read [n|here|all]`
# and use it in his reply. Pull-only — nothing is captured unless he asks,
# and every read is logged.
#
# Needs the extras from requirements.txt plus an OCR engine:
# pip install mss pytesseract && sudo apt install tesseract-ocr
# or, without sudo:
# pip install mss rapidocr-onnxruntime
# mss captures on X11/Windows/macOS but NOT Wayland.
#
# This sends the text of a whole screen to the server when used. That's not a
# new capability — the shell relay could already screenshot and OCR — but it's
# a far easier one to reach for. SCREEN_TEXT=false removes it entirely.
#SCREEN_TEXT=true
#SCREEN_TEXT_MAX_CHARS=4000
# ── Quiet hours / do-not-disturb (optional) ────────────────────────────────
# Comma-separated HH:MM-HH:MM ranges; wrapping past midnight is fine. While
# napping the pet dims, stops wandering and makes no proactive noise — the
@@ -92,6 +190,12 @@ ELEVENLABS_VOICE_ID=
#NOTIFICATION_BRIDGE=false
#NOTIFICATION_FILTER=build|deploy|calendar
#NOTIFICATION_MIN_INTERVAL_SECONDS=60
# Notifications queue while the pet is napping (the heartbeat that forwards them
# doesn't run). These two stop an overnight backlog becoming a monologue at 8am:
# the queue drops its oldest past the limit, and anything staler than the age
# limit is discarded rather than read out.
#NOTIFICATION_QUEUE_LIMIT=20
#NOTIFICATION_MAX_AGE_SECONDS=900
# ── Push-to-talk (optional) ─────────────────────────────────────────────────
# Global hotkey; needs pynput and a session that allows global key hooks
@@ -104,6 +208,25 @@ ELEVENLABS_VOICE_ID=
#WAKE_NEAR_MISS_MARGIN=0.2
#WAKE_NEAR_MISS_LIMIT=40
# ── sudo password prompts (optional) ────────────────────────────────────────
# The pet has no terminal, so a server-relayed `sudo` would block forever on
# a tty nobody is watching. With this on, bare `sudo` becomes `sudo -A` and
# the password is collected in a desktop dialog you have to answer — a real
# askpass binary if one is installed, otherwise a generated zenity/kdialog
# wrapper in ~/.cache/bolt-pet/askpass.sh.
# Set it to false if you'd rather Bolt never be able to ask for root: sudo
# commands then just fail. Read the dialogs — that box is the only thing
# between "Bolt decided to run sudo" and it running.
#SUDO_ASKPASS_PROMPT=true
#SUDO_ASKPASS_HELPER= # blank = auto-detect
#SUDO_COMMAND_TIMEOUT_SECONDS=180 # long enough for a human to answer
# ── File delivery (optional) ────────────────────────────────────────────────
# The server's deliver_files tool ("send me that report") queues workspace
# files on this session; the pet fetches and saves them automatically.
#RECEIVE_FILES=true
#DELIVERED_FILES_DIR=~/Downloads/Bolt
# ── Misc (optional) ──────────────────────────────────────────────────────────
#COMMAND_TIMEOUT_SECONDS=30
#HEARTBEAT_INTERVAL_SECONDS=60
+2
View File
@@ -3,3 +3,5 @@ __pycache__/
*.pyc
.env
.pytest_cache/
.claude
CLAUDE.md
+416 -34
View File
@@ -11,13 +11,19 @@ dependency** on the server repo; it's a standalone HTTP client configured via
its own `.env`.
Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk /
click → record utterance → Deepgram STT → + active-window context → POST
/desk/converse → [server may relay a shell command to run on this machine, or
a `petctl` pseudo-command that moves/emotes the pet instead] → reply →
click → record utterance → Deepgram STT → + active-window + screen-layout
context → POST /desk/converse → [server may relay a shell command to run on
this machine, or a `petctl` pseudo-command that moves/emotes the pet, jumps it
to another monitor, reads a screen's text back, or plays a multi-voice scene
instead] → reply (optionally tagged with a voice the server picked for it) →
ElevenLabs streaming TTS (or offline pyttsx3 fallback) → speakers`, with the
pet sprite/speech bubble reflecting state throughout, and playback
interruptible by talking over it (barge-in).
Because a relayed command's output goes back up the tool-result relay before
the final reply, a `petctl read` mid-turn means Bolt can look at a monitor and
then talk about what's on it in the same answer.
Side channels that let the pet act between turns: the heartbeat (proactive
announcements), the desktop notification bridge, and autonomous wandering —
all suppressed while it's napping (quiet hours / fullscreen DND).
@@ -34,10 +40,19 @@ run.bat # Windows
QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/
.venv/bin/pytest tests/test_state.py::test_happy_path_transitions # single test
# Redraw the pet's sprite frames (the committed PNGs are this script's output)
python scripts/generate_bolt_sprites.py # --out /tmp/x to preview first
# Convert a grid sprite sheet into the per-frame-PNG convention sprite.py expects
python scripts/slice_spritesheet.py path/to/sheet.png assets/sprites/idle --cols 6 --rows 1
```
**Cutting a release:** bump `__version__` in `bolt_pet/__init__.py` in the same
commit you tag, because that string — not the git history — is what every
already-installed pet compares against the newest Gitea tag (`updater.py`). A
tag without the bump means nobody updates; a bump without the tag means the
next tag looks older than what's running.
There is no lint/build step configured beyond pytest. `cp .env.example .env`
and fill in `BOLT_SERVER_URL` / `DESK_API_KEY` (+ `DEEPGRAM_API_KEY`,
`ELEVENLABS_API_KEY`) before running — without server config the controller
@@ -60,42 +75,220 @@ logs a missing-config message and exits its thread instead of starting.
a `QWidget` directly. Also drives the periodic heartbeat
(`_maybe_heartbeat`, gated by `HEARTBEAT_INTERVAL_SECONDS`) which lets the
server push proactive spoken announcements between user turns, and on the
same tick re-evaluates nap state and drains queued desktop notifications.
It owns the live wake-word threshold (`wake_threshold()` is passed to
`listen_for_wake_word` as a *callable* so the tray slider takes effect
mid-listen) and the conversation `history`.
same tick re-evaluates nap state, checks for delivered files, and drains
queued desktop notifications. It owns the live wake-word threshold
(`wake_threshold()` is passed to `listen_for_wake_word` as a *callable* so
the tray slider takes effect mid-listen) and the conversation `history`.
**One thread drives all of it, so failure containment is structural.** Every
entry point that can raise runs inside `_guarded(work, label)`, which logs and
forces the machine back to IDLE (the only state it's always safe to resume
from): the conversation turn, the heartbeat tick — which matters most, since
`on_tick` is the one place control returns to us during a listen that blocks
for minutes, and everything it drives touches the network or shells out — and
the post-restart report. `run()` wraps the lot in try/finally because
`finished` is what `ui/app.py` waits on to quit the thread and to run a
pending `os.execv`; an exception escaping `_loop` used to skip it, so the
failure mode of any bug below was "the pet goes deaf with the mic still open
and the tray won't quit" rather than "one turn failed". `_handle_command` has
the same shape for a different reason: it must **always return a string**,
because the server is blocked on `/desk/tool_result` while it runs and an
exception there means the relay never posts and the server sits out its own
timeout on a turn that can't finish — silent on both ends. Handed back as
command output instead, Bolt can read what broke and say so in the same turn.
- **`server_client.py`** — HTTP client for the desk API, dependency-free
beyond `requests` so it's easy to mock in tests. `converse()` loops relaying
server-issued shell commands (`run_local_command`, executed via
`subprocess.run(shell=True)` as the desktop user, 30s default timeout) via
server-issued shell commands (`run_local_command`, executed via a
`shell=True` `Popen` as the desktop user, 30s default timeout) via
`/desk/tool_result` until the server sends a final `reply` (capped at
`_MAX_RELAY_HOPS`). This is the same "full desktop control" trust model as
`_MAX_RELAY_HOPS` — exhausting which is reported as its own error, because
"unknown server response" sent everyone looking at the payload shape when what
happened is a model that kept calling tools and never answered).
`run_local_command` is `Popen` rather than `subprocess.run` for the timeout
path: the command is a shell, and `run()`'s timeout would kill only that
shell, leaving whatever it spawned (a build, a `tail -f`, an ffmpeg) alive for
the rest of the session with no parent watching — so the child gets its own
process group (`start_new_session`, POSIX) and a timeout SIGTERMs the group,
SIGKILLs it two seconds later, then drains the pipes *with its own timeout* so
a grandchild holding stdout can't turn a timeout into a hang. Whatever the
command printed before it hung is returned alongside the timeout notice, since
the last line usually says exactly what it was stuck waiting for. This is the
same "full desktop control" trust model as
the server repo's other desk clients — commands only ever originate from
the user's own voice/click requests in their own session.
the user's own voice/click requests in their own session. A final reply is
returned as a `Reply(text, voice_id, voice_name)` rather than a bare string,
because the server can tag it with a voice — see "Voices" below.
`list_outbox_files`
/ `download_outbox_file` hit the same `/desk/files` and `/desk/files/<id>`
endpoints the server's `deliver_files` tool queues onto — see `file_delivery.py`.
- **`file_delivery.py`** — the filesystem half of receiving files the server
queues via its `deliver_files` tool (`ai/desk_api.py` in the main tmn-api
repo — "send me that report" during a conversation spools the matched
workspace files, zipping multiple into one, onto the session's outbox).
`controller._check_deliveries` lists `/desk/files` and downloads anything
queued — right after a conversation/notification turn (the common case)
and once per heartbeat tick for anything queued out-of-band — saving each
under `DELIVERED_FILES_DIR` (default `~/Downloads/Bolt`). Downloading a
file dequeues it server-side, so it's only ever handed out once; `save()`
never overwrites an existing download, suffixing `" (1)"`, `" (2)"`, ... on
a name collision. `sanitize_filename()` reduces a server-supplied name to
its bare filename (`Path(...).name`), which is defense-in-depth against a
delivered name that's secretly a path, since a per-user desk API key means
the name isn't always coming from someone as trusted as the owner. Toggle
off entirely with `RECEIVE_FILES=false`.
- **`audio/`** — `mic.py` (energy-based VAD utterance capture, ported from the
server repo's `bolt_desk.py`), `wake_word.py` (openWakeWord `thunderbolt.onnx`
server repo's `bolt_desk.py`, plus `flush()` — see the note below on the pet
hearing itself), `wake_word.py` (openWakeWord `thunderbolt.onnx`
detection + `NearMissLog` for threshold tuning — see below), `stt.py`
(Deepgram), `tts.py` (ElevenLabs, streaming by default — `stream_pcm()` +
`play_stream()` start playback on the first chunk; `chunks_to_int16()`
carries odd bytes across HTTP chunk boundaries, without which everything
after the first split sample plays as static — falling back to whole-clip
PCM then offline `pyttsx3`), `barge_in.py` (`BargeInDetector`: N consecutive
loud mic frames while the pet is talking cuts playback and starts the next
turn; threshold is deliberately ~4x the VAD one because the mic hears the
pet's own voice). Each accepts an injectable stream/model/protocol so tests
don't need real audio hardware or a display.
PCM then offline `pyttsx3`; every entry point takes an optional `voice_id`
overriding `ELEVENLABS_VOICE_ID`, and `model_for()` picks the multilingual
model whenever there's an override or non-ASCII text, since the default
`eleven_flash_v2` is English-only and would read either as garbled
phonetic English rather than failing), `barge_in.py` (two detectors behind one
`reset()`/`check()` shape, chosen by `BARGE_IN_MODE` via `make_detector`:
**wake** (default) scores every frame with the same openWakeWord model the
idle listener uses, so only the wake phrase cuts playback; **energy** is the
original N-consecutive-loud-frames rule, threshold ~4x the VAD one because
the mic hears the pet's own voice. Wake mode shares `_default_model` with
the idle listener — the two never run concurrently — and `reset()`s it on
detection so the tail of one reply can't count toward the next). Each
accepts an injectable stream/model/protocol so tests don't need real audio
hardware or a display.
- **`pet_actions.py`** — `petctl` pseudo-commands (`petctl move top-left`,
`petctl emote wave`, `say`/`wander`/`nap`). The desk API has no "move the
pet" payload type and this repo can't change the server, so these ride the
existing shell-command relay: `controller._handle_command` parses them and
they never reach `subprocess`; anything else is a real shell command exactly
as before. Pure parsing; the UI half is `PetWindow.apply_action`.
`petctl emote wave`, `say`/`wander`/`nap`, the screen verbs
`jump`/`monitors`/`read`, and `voice reset`). The desk API has no "move the pet" payload type
and this repo can't change the server, so these ride the existing
shell-command relay: `controller._handle_command` parses them and they never
reach `subprocess`; anything else is a real shell command exactly as before.
Pure parsing; the UI half is `PetWindow.apply_action`. Note `jump`'s target
is *not* validated here — which monitors exist is a runtime fact this pure
module doesn't have, so the spec passes through to `monitors.resolve()`.
Query verbs (`monitors`, `read`) are answered in `_handle_command` rather
than by `pet_actions.describe()`, because their output *is* the point: it
goes back up the tool-result relay for Bolt to use in his reply — as is
`voice reset`, which reports what it dropped since the server can't see
which voice is in use. `voice` only ever resets: picking one is the
server's job (`speak_as`, which it already knows how to use), so a
`petctl voice <name>` attempt is an error pointing back at that marker.
- **`self_restart.py`** — `petctl self_restart`, the pet restarting itself so
Bolt can *see* a code change he just made instead of waiting for a human to
restart it. Three problems shape it, and all three are the interesting part.
(1) The restart can't happen inline: killing the process mid-turn would drop
the HTTP tool relay before the result was posted, leaving the server to wait
out its timeout on a turn that can never finish — so the command only
*arms* it (`controller._arm_self_restart`) and
`controller._maybe_self_restart` fires it after the reply is spoken, the
same "only between turns" rule the updater follows. (2) A broken edit must
not be fatal, so `preflight()` imports the package in a **subprocess**
before arming — this process holds the old modules, so an in-process import
would pass on a file that no longer parses — and a SyntaxError comes back as
the command's output, in the same turn, with the pet still running. (3) The
reason has to outlive the process, so it's written to
`~/.cache/bolt-pet/restart_context.json` (never inside the repo Bolt is
editing) and read on the way back up by `controller._report_self_restart`,
which posts it to the server as an ordinary turn — that's what makes
"restart and check the sprites load" finish as a spoken sentence rather than
a silence. `check_loop_guard` refuses after `SELF_RESTART_MAX` restarts in
`SELF_RESTART_WINDOW_SECONDS`, so an edit-restart-crash cycle stops itself.
Off switch: `SELF_RESTART=false`.
- **`dialogue.py`** — `dialoguectl` pseudo-commands: a multi-voice *scene*
through ElevenLabs' Text to Dialogue endpoint (`audio/tts.
synthesize_dialogue`), checked in `_handle_command` between petctl and
filectl. Same single-line-JSON wire format as filectl and for the same
reason (the server's `command` marker captures only up to the next
newline), and it accepts the ElevenLabs field names (`inputs`/`voice_id`)
as well as its own (`lines`/`voice`) because the model has read that API
and copying its shape is the obvious thing to try. Voices are *named*
(`DIALOGUE_VOICES` maps names to ids) rather than pasted as raw ids, and
`self` resolves to whatever voice the pet is speaking with right now —
including a `speak_as` pick — so Bolt sounds like himself in his own
scenes. The API's limits (10 distinct voices, ~2000 characters) are
enforced *before* the request so a mistake comes back up the tool-result
relay as a sentence Bolt can act on rather than an HTTP 422 he can't see.
Unlike the normal reply path there is no streaming variant, so a scene is
whole-clip: `controller._play_dialogue` plays it with the same bubble,
transcript and barge-in handling a spoken reply gets, and returns to
THINKING afterwards (not IDLE) because the server is still waiting on the
tool result — that leg is why `state.py` allows TALKING -> THINKING.
- **`file_ops.py`** — `filectl` pseudo-commands, checked in `_handle_command`
right after petctl and before falling through to a real shell command.
Executing arbitrary commands already worked via the shell relay
(`run_local_command` — see `server_client.py` below); what filectl adds is
a *reliable* way to do the read/write/edit/list slice of that, since
getting the model to hand-roll a shell heredoc for multi-line content full
of quotes/`$`/backticks is failure-prone — and `list` exists as its own op
(rather than relying on the model shelling out to `ls`/`dir`) because this
project is cross-platform and the model shouldn't have to guess which
listing command applies on Windows vs. Linux vs. macOS; one glob-based op
(`pattern`, default `*`; `recursive` for `rglob` instead of `glob`) covers
all three. Wire format is `filectl <json>` where `<json>` is a
**single-line** compact JSON object —
`{"op": "list"|"read"|"write"|"edit", "path": ..., ...}` — not a multi-line
marker block (an earlier design): the server relays this as the argument
to the ordinary `command` tool marker, and that marker's extractor
(`ai/agents/default.py` in the main repo) only captures up to the next
newline, so anything genuinely multi-line silently got truncated no matter
how the prompt worded it. JSON sidesteps that for free — `json.dumps`
already encodes embedded newlines as the two characters `\n`, not a real
line break, so multi-line file content still fits on the one physical line
the extractor sees. `edit` requires the old text to match exactly once —
same discipline as this project's own code-editing tool — and raises
rather than guessing if it's missing or ambiguous. This doesn't expand
what the server can do to this machine (a relayed shell command could
already overwrite anything the desktop user can write — see the security
notes below); it's a safer path to the same capability. Pure parsing
(`parse`) is separated from the filesystem I/O (`execute`), matching
pet_actions.py's parse/describe split.
- **`relay_json.py`** — the JSON parser both `filectl` and `dialoguectl` use
instead of `json.loads`, because their payload is hand-typed by a model into
a tool marker and fails in a small, repeatable set of ways (stray quote after
a bare literal, trailing comma, single or smart quotes, Python `True`/`False`,
a markdown fence). Strict parsing already cost a live turn: the call was
rejected, the model re-sent the identical line, was rejected again, and then
told the user "I'll check now" without ever calling anything. So `loads()`
tries strict first, then applies **named, individually-narrow repairs** and
accepts one only if the result parses — and on total failure raises
`RelayJsonError` carrying a caret pointed at the offending character, since
a model can act on a pointed-at fragment but not on "Expecting ',' delimiter:
char 74". Two conventions matter for any new relayed-JSON command: repairs
are **never silent**`parse` stashes them on the action as `_repairs` and
`describe` appends `relay_json.repair_note(...)` to the tool result, so the
model is told it sent something broken while it still has the turn — and new
repairs go in the `_REPAIRS` tuple ordered cheapest/safest first. Tested
inside `tests/test_file_ops.py`, not a file of its own.
- **`screen_context.py`** — active-window title (xprop/xdotool, Win32,
osascript) appended to each utterance via `context_for()`, plus
`is_fullscreen_active()` for do-not-disturb. Text only — the desk API takes
no images. Every probe is best-effort and returns None/False rather than
raising; the parsing is split into pure functions that are tested without a
display server.
- **`monitors.py`** — the screen layout, and resolving `petctl jump` targets
(a 1-based number, a name, `next`/`prev`/`primary`/`other`, or a direction
like `left`/`up` worked out from the actual geometry). Pure — no Qt, no
subprocess. The monitor list is *published by the UI*
(`PetWindow.publish_monitors` builds it from `QGuiApplication.screens()` and
emits it over a queued signal to `controller.set_monitors`), because the
controller and the window must agree on what "monitor 2" means: enumerating
with `xrandr` on one side and Qt's screen list on the other gives different
orderings on the same machine, and Bolt would announce one screen and land
on another. Qt is the single source of truth; `Monitor.index` is 0-based and
`.number` is the 1-based value used in every string a human or the model
sees. The controller resolves a jump to a concrete index *before* emitting
it, so the window can't re-resolve against a different list.
- **`screen_text.py`** — OCR, so Bolt can read what's on a monitor
(`petctl read [n|here|all]`). **Pull, not push**: nothing captures on its own
— the server has to ask, and the text goes back as that command's output.
That's deliberate; OCR of a 4K screen costs a second or two that would
otherwise be added to *every* utterance, and screen contents leaving the
machine should be a visible decision rather than a constant. Capture needs
`mss` (X11/Win32/macOS, **not** Wayland), recognition needs Tesseract or
RapidOCR; both are optional and soft-fail with a reason the way `hotkey.py`
does, and `read_monitor()` never raises because its return value is command
output. Engine selection takes injected probes so it's testable wherever.
- **`quiet.py`** — quiet-hours spec parsing (`23:00-08:00`, wraps midnight,
comma-separated). Napping suppresses *proactive* noise and wandering only;
wake word / click / push-to-talk still work.
@@ -103,6 +296,41 @@ logs a missing-config message and exits its thread instead of starting.
`dbus-monitor`, parses Notify calls (pure `iter_notifications()`), filters
and rate-limits them (`NotificationGate`), and the controller forwards
survivors through `converse()`. Off by default — each one is a round trip.
Note where the queue between the two threads lives: notifications arrive on
the watcher thread and are forwarded from the heartbeat, which **doesn't run
while the pet is napping** — so they accumulate overnight. The controller's
queue is therefore a bounded `deque` stamped on arrival, and the drain
discards anything older than `NOTIFICATION_MAX_AGE_SECONDS` rather than
reading a nine-hour-old backlog out at 8am. A drain that stops early (a nap
starting mid-loop, or the server going down) re-queues what it didn't forward
instead of dropping it, which the original swap-and-return did silently.
- **`sudo_askpass.py`** — makes server-relayed `sudo` usable from a process
with no terminal, by pointing sudo's `SUDO_ASKPASS` at a GUI helper and
rewriting bare `sudo` to `sudo -A` (`add_askpass_flag`, a conservative regex
that skips anything already carrying a flag and anything inside quotes).
Prefers a real askpass binary and falls back to generating a
zenity/kdialog wrapper in `~/.cache/bolt-pet/askpass.sh`. Resolution order
is injectable (`is_executable`/`which`) so it's testable on a machine with a
different set installed. See the security notes — the dialog is the boundary.
- **`updater.py`** — self-update from the Gitea releases API. Polls
`<UPDATE_REPO_API>/releases/latest` for a tag newer than
`bolt_pet.__version__` and moves the checkout to it with
`git fetch --tags` + `git checkout tags/<tag>`, so "downloading an update"
is just git and rolling back is one command. Three safety rules: a **dirty
working tree is skipped, never stashed** (silently discarding your
work-in-progress beats running an old version); everything after the
checkout — dependency install, then an **import smoke test in a
subprocess** (this process still has the old modules loaded, so importing
in-process would prove nothing) — is guarded, and any failure rolls back to
the exact ref that was live before, branch name or SHA; and the restart only
happens once the new code imports, so a broken release costs a log line
rather than a pet that won't start. Git goes through an injectable
`run(args) -> (code, output)` callable so apply/rollback is unit-tested
against a fake git; version comparison and release parsing are pure.
`controller._maybe_update` drives it from the wake-listener tick (so the pet
is IDLE and between turns by construction) and the actual `os.execv` happens
in `ui/app.py` *after* `app.exec()` returns — that ordering is what
guarantees the mic is released before the new process opens it.
- **`history.py`** — rolling transcript (`HISTORY_LIMIT` turns) behind the
tray's History window and click-to-copy on the bubble.
- **`hotkey.py`** — global push-to-talk via `pynput`; soft-fails with a logged
@@ -112,9 +340,43 @@ logs a missing-config message and exits its thread instead of starting.
`for_speech()` (called inside `tts.speak()`, so every path to the speakers is
covered) strips markdown, emoji, URLs and stray symbols the voice would read
literally ("asterisk asterisk"), turns bullet lists into full sentences, and
words a few symbols (`&` → "and"). `for_display()` is the looser version for
the speech bubble — markdown syntax gone, emoji kept. Pure string logic, no
words a few symbols (`&` → "and"), abbreviations the voice would spell out
letter by letter (`e.g.` → "for example", `etc.` → "and so on") and a long
option's leading `--` (heard as "dash dash force"; the single hyphen has to
survive for "bolt-pet"). `for_display()` is the looser version for
the speech bubble — markdown syntax gone, emoji kept. `is_question()` decides
whether a reply leaves the pet waiting on an answer: it tests the *spoken*
form (so a '?' inside a stripped code block or URL doesn't count) and a '?'
**anywhere** counts. That last part was once trailing-only, on the theory that
"What time is it? It's 7:15." isn't awaiting a reply — true of that sentence
and wrong more often, since Bolt routinely asks and then keeps talking ("Want
me to fix it? I'd start with the config"), which is the case that actually
costs you a wake word. The asymmetry is the argument: an unwanted extra listen
ends itself on `VAD_GRACE_SECONDS` of silence, a missed one makes you start
over. `controller._should_follow_up` uses it to keep listening without the
wake word, capped by `FOLLOW_UP_MAX_TURNS` so a server that ends every reply
with a question can't loop forever off mic noise. Pure string logic, no
Qt/audio imports.
- **`intents.py`** — the handful of utterances answered *without* the server.
"stop", "come here", "go to sleep", "say that again", "use your normal voice"
are commands to the body, and routing them through the desk API costs two to
four seconds and three network hops to make the pet walk left — and only works
if the server's prompt happens to advertise the matching `petctl` verb (which
is why `voice reset` needs a block in `ai/desk_api.py`'s pet prompt; see the
Voices section). Recognising the phrase here removes both the latency and that
coupling. The design problem is *not stealing real requests*, and three rules
cover it: whole-utterance exact match after normalisation (so "stop" is an
intent and "stop the docker container" is a question for Bolt), a closed table
with nothing arguable in it, and **never on a follow-up turn** — if Bolt just
asked you something your answer is his, and swallowing "never mind" locally
would leave the server holding a question it never got an answer to. Both
sides of the comparison go through `normalize()` (the table is canonicalised
at import, and `_build()` refuses to build one where two intents claim the
same normalised phrase, or where a phrase reduces to "" and would match pure
filler like "hey bolt"). Actions come back in the **same shape
`pet_actions.parse` produces**, so `PetWindow.apply_action` needs no new
vocabulary; the effects live in `controller._handle_local_intent`. Off switch:
`LOCAL_INTENTS=false`.
- **`ui/`** — `app.py` wires `QApplication` + `PetWindow` + `PetTray` + the
history/tuner windows + the push-to-talk hotkey + the controller thread
together; `pet_window.py` is the frameless/translucent/always-on-top sprite
@@ -124,6 +386,15 @@ logs a missing-config message and exits its thread instead of starting.
target every `PET_WANDER_INTERVAL_SECONDS` (randomized), suppressed
whenever the pet is non-IDLE, napping, dragged, or has a bubble up. A
commanded `petctl move` overrides all of that except the drag.
- **the walk cycle** — while actually travelling, `_animation_key()` swaps
the state animation for the side-view `walk/` frames (not a `PetState`
see the sprites README). It is stepped by *distance travelled*
(`_WALK_PIXELS_PER_FRAME`), never by the animation timer, so the planted
paw tracks backwards at exactly the speed the window moves forwards;
`_advance_frame` deliberately no-ops while walking so the two can't
double-step it. The art is drawn facing right and `_oriented()` mirrors it
(cached per frame) when heading left. No `walk/` art → falls back to the
old coded bob rather than a placeholder blob.
- **emotes** — `emote_transform()` is pure maths (dx, dy, rotation, scale
from a 0..1 progress) kept out of `paintEvent` so the curves are unit
tested; every emote must return to the identity transform at progress 1.0
@@ -136,12 +407,46 @@ logs a missing-config message and exits its thread instead of starting.
- **edge snapping** (`PET_EDGE_SNAP`) after a drag or a stroll, and **nap
dimming** (`set_napping`).
`sprite.py` loads `assets/sprites/<state>/*.png` (filename-sorted, looping —
currently Kenney's CC0 robot pack, see `assets/sprites/README.md`) and falls
the art is *generated* by `scripts/generate_bolt_sprites.py`, a Pillow
drawing of Bolt as a shepherd pup; edit the script and re-run it rather than
the committed PNGs, see `assets/sprites/README.md`) and falls
back to a procedurally-drawn placeholder blob per state if a folder has no
frames; `tray.py` is the system tray menu (talk now / mute / nap / wander /
click-through / history / wake-word tuning / quit) — the pet window has no
title bar or taskbar entry; `history_window.py` and `wake_tuner.py` are the
two dialogs it opens.
frames. It also loads `EXTRA_ANIMATIONS` — currently just `walk/` — keyed by
name rather than by `PetState`, with `has()` reporting whether a key is
backed by real art so callers can decline a placeholder instead of trotting
a blob across the desktop; `tray.py` is the system tray menu (talk now / mute / nap / wander /
click-through / history / wake-word tuning / use-default-voice / quit) — the
pet window has no title bar or taskbar entry; `history_window.py` and
`wake_tuner.py` are the two dialogs it opens.
### Voices (the server's `speak_as`)
Ask Bolt to talk like someone else, or in another language, and the *server*
does the picking: its desk-only `voice_search` marker browses the ElevenLabs
voice library, and `speak_as: <voice_id>` on the final reply tags that reply
with the chosen voice (adding a Voice Library pick to the ElevenLabs account
first, so the id is usable by the time it reaches us). Nothing about that is
this repo's to decide — all the client owes it is actually speaking in the
voice it was handed: `converse()` returns it on `Reply`, `_apply_voice()`
records it, and `_speak()` passes it to `tts.speak(voice_id=...)`.
Two things are decided *here*, though, because the server can't:
- **The voice sticks** (`VOICE_STICKY`, default on). The server tags one
reply and strips the marker before storing the turn, so it never sees the
id again — "keep talking like that" would send it searching for a voice all
over again, and it'd likely land on a different one. Holding the id
client-side is what makes the rest of the conversation stay in that voice.
An untagged reply therefore never *changes* the voice; only a new
`speak_as`, `VOICE_STICKY=false`, or a reset does.
- **There's a way back.** Since the server was never told Bolt's own voice
id, it can't ask for it back with `speak_as` — so reverting is local: the
tray's **Use default voice** entry (enabled only while a picked voice is
in use, kept in sync by the `voice_changed` signal), a restart, or
`petctl voice reset`, which is what lets Bolt honour "go back to your
normal voice" out loud. That last one needs the server's pet prompt block
(`ai/desk_api.py`, `pet_tools`) to mention the verb, or the model never
emits it — the desk API's prompt is where petctl is advertised.
### Wake-word detection
@@ -155,6 +460,22 @@ detection. Swap `WAKE_MODEL_FILE` to point at a differently-trained `.onnx`
model to change the wake phrase — everything downstream (STT, server call,
TTS) is unaffected.
**openwakeword's `Model.reset()` is not enough to forget a detection.** It
clears the *prediction* buffer only; the rolling audio window the classifier
actually scores lives in `model.preprocessor` (`raw_data_buffer` — 10s of raw
audio — plus `melspectrogram_buffer` and a ~120-frame `feature_buffer`) and
`AudioFeatures` has no reset method at all. So after a detection the wake
phrase is still in the window, and the next frame fed to the model re-fires on
it. Symptom when this bites: the pet cuts itself off a word into every reply,
because wake-mode barge-in resumes feeding the model and instantly matches the
"thunderbolt" that *started* the turn. `wake_word.hard_reset(model)` restores
the preprocessor to its as-constructed (silence) state and is what both
`listen_for_wake_word` and `WakeWordBargeIn.reset()` call — use it, not
`reset()`, anywhere a detection needs to be genuinely forgotten. The blank
state is cached on the preprocessor object (not in an `id()`-keyed dict —
CPython reuses ids after GC), since rebuilding it costs an ONNX pass over 10s
of silence.
The threshold is tunable at runtime: the tray's **Wake word tuning…** window
(`ui/wake_tuner.py`) shows the peak score seen and a rolling list of near
misses (frames within `WAKE_NEAR_MISS_MARGIN` *below* the threshold — i.e.
@@ -162,6 +483,29 @@ the times it nearly heard you), and its slider is read per frame because
`listen_for_wake_word` accepts a callable threshold. Set the threshold just
under the peak you can hit reliably, then persist it in `.env`.
### The mic keeps recording while nothing is reading it
Same family of bug as the openwakeword one above, one layer down: PortAudio
captures into a ring buffer continuously, so audio from a stretch where the
pipeline thread was busy elsewhere is still queued when the next read happens.
It bites in exactly one place. At the end of a reply that asked you something,
`_speak` sets `_talk_now` and the next turn starts recording immediately — with
the tail of the pet's own TTS sitting in that buffer, above the VAD threshold.
The VAD takes it for the start of your answer, Deepgram transcribes it, and Bolt
is handed his own last sentence as if you had said it. With barge-in on the
detector was draining the stream during playback so the window is small; with
`BARGE_IN=false` nothing drains it at all.
`mic.flush(stream)` drops what's buffered, and `_speak` calls it on the
follow-up branch only. **That placement is the whole correctness argument**
flushing is only safe where the buffer is known to hold nothing *you* said:
playback ran to completion, so if you had spoken, barge-in would have cut it and
taken the interrupted branch instead. Never flush before a wake-triggered
recording, where the rest of "thunderbolt, what time is it" is legitimately
queued and dropping it clips the request. A single call is bounded by
`max_seconds` so it can't chase a stream filling as fast as it drains, and it
no-ops on a stream with no `read_available` (i.e. every fake stream in tests).
### Testing conventions
`tests/` covers pure logic only (state machine, wake-word scoring loop, mic
@@ -191,9 +535,47 @@ matches the trust model of the server repo's other desk clients. Keep
`DESK_API_KEY` private and don't expose the desk API port to the open
internet.
Two newer features widen what leaves this machine, both switchable in `.env`:
`file_ops.py`'s `filectl` read/write/edit pseudo-commands ride that same
relay and are bound by the same trust model — no path is off-limits beyond
normal filesystem permissions for the desktop user, exactly like a relayed
`cat`/`sed`/`rm` already isn't. They don't grant the server anything a shell
command couldn't already do; they just make the read/write/edit path
reliable instead of relying on the model getting shell quoting right.
`sudo_askpass.py` widens that further, by design: with `SUDO_ASKPASS_PROMPT`
on (the default), a relayed bare `sudo` is rewritten to `sudo -A` and the
password is collected in a desktop dialog, so commands can escalate to root
instead of hanging on a tty the pet doesn't have. The dialog is the security
boundary — it's the only thing between the server deciding to run `sudo` and
it running, so the prompt is deliberately not suppressible per-command and
those commands get their own longer timeout (`SUDO_COMMAND_TIMEOUT_SECONDS`)
rather than being made non-interactive. Set `SUDO_ASKPASS_PROMPT=false` to
take the capability away entirely; sudo commands then fail. Note that
`sudo -n` / `sudo -A` / `sudo -u …` in a relayed command are never rewritten,
so an explicit non-interactive sudo stays non-interactive.
**Don't run the pet as root.** It needs no privileges of its own, PortAudio
can't reach the user's PipeWire socket from a root session (raw ALSA devices
reject the 16 kHz capture rate — `paInvalidSampleRate`), and every relayed
command would run unconstrained.
Several features widen what leaves this machine, all switchable in `.env`:
`SCREEN_CONTEXT` appends the focused window's *title* to each utterance
(titles often contain file paths, document names, or subject lines), and
`NOTIFICATION_BRIDGE` (off by default) forwards matching desktop
notifications to the server. Neither sends screenshots or notification
contents you haven't matched with `NOTIFICATION_FILTER`.
(titles often contain file paths, document names, or subject lines),
`MONITOR_CONTEXT` appends the screen layout (sizes and names only — no
contents), and `NOTIFICATION_BRIDGE` (off by default) forwards matching
desktop notifications to the server. None of those send screenshots or
notification contents you haven't matched with `NOTIFICATION_FILTER`.
`SCREEN_TEXT` is the biggest of them: `petctl read` OCRs a whole monitor and
sends the recognised text to the server — everything visible, not just the
focused window. Two things keep it honest. It's **pull-only**: no capture
happens unless the server explicitly asks, so it can't leak in the background
the way a per-turn annotation would, and each read is logged. And it is
strictly *not* a new capability — the shell relay could already run a
screenshot tool and pipe it through OCR — it just makes a thing the trust
model already allowed reliable, bounded (`SCREEN_TEXT_MAX_CHARS`) and
visible. It is nonetheless far easier to reach for than the shell route, so
if that trade isn't one you want, `SCREEN_TEXT=false` removes it and
`petctl read` starts reporting that it's disabled. Capture is `mss`-based and
therefore silently unavailable on Wayland.
+39 -4
View File
@@ -69,8 +69,8 @@ limitations).
speaks, and barging in starts your next turn immediately (`BARGE_IN`).
- Right-click the tray icon for **Talk now**, **Mute mic**, **Nap**,
**Wander around**, **Click through the pet**, **History…**, **Wake word
tuning…** and **Quit** — the pet window itself has no title bar or taskbar
entry.
tuning…**, **Use default voice** and **Quit** — the pet window itself has
no title bar or taskbar entry.
- **Click the speech bubble** to copy what it just said; the tray's
**History…** window keeps the last `HISTORY_LIMIT` turns.
@@ -80,8 +80,8 @@ limitations).
listening/thinking/talking or while a bubble is up.
- **Moves and emotes on command.** Bolt can relay `petctl move top-left`,
`petctl emote wave|hop|spin|nod|shake`, `petctl say ...`, `petctl wander
on|off`, `petctl nap on|off`. These are intercepted here and never reach a
shell.
on|off`, `petctl nap on|off`, `petctl voice reset`, and `dialoguectl` for a
multi-voice scene. These are intercepted here and never reach a shell.
- **Naps** during `QUIET_HOURS` (e.g. `23:00-08:00`) or while a fullscreen
app is focused (`DND_ON_FULLSCREEN`) — it dims, stops wandering, and makes
no proactive noise. It still answers when you speak to it.
@@ -90,6 +90,41 @@ limitations).
notifications get forwarded to the server, so it can tell you the deploy
went green. Off by default: each one costs a round trip.
## Speaking in another voice
Ask for a different voice — "use a clearer voice", "talk like a pirate", "say
that in Japanese" — and Bolt searches the ElevenLabs voice library on the
server, picks one, and tags his reply with it (`speak_as`); the pet is what
actually speaks in it. A Voice Library pick is added to your ElevenLabs
account automatically the first time it's used, and non-English replies (or
any picked voice) go through `ELEVENLABS_MULTILINGUAL_MODEL_ID` rather than
the English-only `eleven_flash_v2` default.
The new voice **stays on** for the rest of the conversation, because the
server tags a single reply and doesn't remember which voice it chose — so
"keep talking like that" would otherwise send it hunting for a voice again.
To get his own voice back: ask him ("use your normal voice" — he relays
`petctl voice reset`), use **Use default voice** in the tray menu (greyed
out unless a picked voice is active), or restart the pet. Set
`VOICE_STICKY=false` in `.env` if you'd rather each pick lasted exactly one
reply.
## Multi-voice dialogue
Ask for a scene — "do the argument between the two of them", "read that back
as a radio play" — and Bolt can relay a `dialoguectl` command that the pet
renders through ElevenLabs' Text to Dialogue endpoint: several voices in one
take, with delivery tags the v3 model acts on (`[cheerfully]`, `[whispering]`,
`[stuttering]`). One request per scene, so the voices actually react to each
other instead of sounding like clips glued together.
Name the cast in `.env` (`DIALOGUE_VOICES=narrator:9BWts…,villain:IKne3…`);
the name `self` always means whatever voice the pet is currently using, so
Bolt sounds like himself in his own scenes — including after a `speak_as`
switch. Scenes show up in the speech bubble with the tags stripped, count as
normal speech for the transcript, and can be talked over like any other reply.
`DIALOGUE=false` turns the whole thing off on this device.
## Wake-word detection
`bolt_pet/audio/wake_word.py` feeds every mic frame into `thunderbolt.onnx`
+7
View File
@@ -0,0 +1,7 @@
"""Bolt desktop pet.
__version__ is what the auto-updater compares against the newest tag on the
Gitea releases page (see updater.py), so bump it in the same commit you tag.
"""
__version__ = "0.2.3"
+64 -17
View File
@@ -1,37 +1,84 @@
# Sprite assets
Art: [Kenney's Robot Pack](https://kenney.nl/assets/robot-pack) (CC0 — no
attribution required, credited here anyway), the green side-view robot.
Source pack lives at `~/Documents/kenney_robot-pack`; only the frames listed
below were copied in.
Art: Bolt himself — a cream shepherd pup with a slate cap, a lightning blaze
on his forehead and a bolt tag on his collar. The frames are **generated, not
hand-drawn**: `scripts/generate_bolt_sprites.py` draws every one of them with
Pillow and writes this folder.
```bash
python scripts/generate_bolt_sprites.py # rewrite this folder
python scripts/generate_bolt_sprites.py --out /tmp/prev # preview elsewhere first
python scripts/generate_bolt_sprites.py --states idle # just one state
```
That means tweaking the art is editing code, not 24 PNGs: the palette is a
block of constants at the top of the script, the body/head/ear/tail shapes are
one function each in normalised 0..1 coordinates, and each state's animation is
a list of pose dicts in `frames_for()`. Everything is super-sampled 4x and
downscaled on save, because PIL's draw primitives have no antialiasing.
**Regenerate after editing** — the PNGs here are committed, so a change to the
script alone doesn't move the pet.
Convention the loader (`bolt_pet/ui/sprite.py`) expects:
```
assets/sprites/
idle/ frame_00.png robot_greenBody (standing)
listening/ frame_00.png, frame_01.png robot_greenDrive1/2 (tracks rolling — "leaning in")
thinking/ frame_00.png, frame_01.png robot_greenDamage1/2 (flicker — "processing")
talking/ frame_00.png, frame_01.png robot_greenBody, robot_greenJump (bounce)
error/ frame_00.png robot_greenHurt
idle/ frame_00..07.png breathing, tail wag, blink on frame 06
listening/ frame_00..03.png ears perked, head tilted in, collar tag lit, sound arcs
thinking/ frame_00..05.png eyes up, head cocked, cycling dots
talking/ frame_00..03.png mouth open/close with tongue, ears bouncing
error/ frame_00..01.png X eyes, ears drooped, red spark
walk/ frame_00..07.png side-view walk cycle (see below)
```
- One subfolder per pet state (matches `bolt_pet.state.PetState`).
- One subfolder per pet state (matches `bolt_pet.state.PetState`), **plus
`walk/`**, which is not a state — see below.
- Any `*.png` filenames work — they're played back in alphabetical-sort
order, looping, at `IDLE_ANIMATION_FPS` (see `.env`).
- Frames are scaled to fit within `PET_SIZE` (default 160px), keeping aspect
ratio, and centered in the (square) pet window — the source art here isn't
square, so don't assume it fills the frame edge-to-edge.
order, looping, at `IDLE_ANIMATION_FPS` (see `.env`). At the default 6fps
the 8-frame idle loop runs about 1.3s.
- Frames are square (320px, 2x the default `PET_SIZE` of 160) so they
downscale cleanly; the loader scales to fit `PET_SIZE` keeping aspect ratio
and centres them in the square pet window.
- A state directory with no frames in it falls back to a small
procedurally-drawn placeholder blob (see `_placeholder_frames` in
`sprite.py`).
## The walk cycle
`walk/` is the one animation that isn't a `PetState`. Walking is a property of
*movement* — orthogonal to whether he's idle, listening or talking — so it
stays out of the state machine and is keyed by name instead
(`sprite.EXTRA_ANIMATIONS`). `PetWindow` uses it whenever the pet is actually
travelling and falls back to the state animation the moment it stops.
Three things about it are load-bearing if you redraw it:
- **It's a side view, drawn facing right.** The other poses are a
front-facing sit, which is fine standing still but slides like a chess
piece when moving. `PetWindow._oriented()` mirrors the frames (cached) when
he walks left, so only the right-facing version exists on disk.
- **The cycle is advanced by distance travelled, not by the animation
timer** (`_WALK_PIXELS_PER_FRAME`, one frame per ~13px). That's what keeps
a planted paw tracking backwards at exactly the speed the window moves
forwards. Drive it off the clock and the feet skate whenever
`PET_WANDER_SPEED` doesn't happen to match `IDLE_ANIMATION_FPS`. If you
change the number of frames or the stride length in
`paw_position()`, retune that constant to match or he'll moonwalk.
- **The frames carry their own vertical bob**, so the window's own bob is
switched off while they're in use. Only the no-walk-art fallback still
bobs in code.
Delete `walk/` and everything still runs — he reverts to sliding with a small
coded bob, which is what the pet did before the cycle existed.
## Swapping in different art
Replace any state's PNGs (same alphabetical-order-loops convention) to
change its look — no code changes needed. If your source is a single grid
spritesheet (rows/cols of frames in one PNG) rather than one-file-per-frame,
use `scripts/slice_spritesheet.py` to cut it into this folder-of-frames
change its look — no code changes needed, and nothing forces you to keep
using the generator. If your source is a single grid spritesheet (rows/cols
of frames in one PNG) rather than one-file-per-frame, use
`scripts/slice_spritesheet.py` to cut it into this folder-of-frames
convention:
```bash
Binary file not shown.

Before

Width:  |  Height:  |  Size: 5.6 KiB

After

Width:  |  Height:  |  Size: 55 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 54 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.4 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 57 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 59 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 56 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.3 KiB

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.3 KiB

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.4 KiB

After

Width:  |  Height:  |  Size: 59 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 4.0 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.7 KiB

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.0 KiB

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 61 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 61 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 46 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 46 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 43 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 43 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 46 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 45 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 42 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 43 KiB

+149 -21
View File
@@ -1,26 +1,52 @@
"""Barge-in: notice that the user started talking *while the pet is talking*
so playback can be cut short mid-sentence.
"""Barge-in: notice that the user wants to interrupt *while the pet is
talking* so playback can be cut short mid-sentence.
Deliberately dumber than the utterance VAD in mic.py. The mic hears the pet's
own voice coming back out of the speakers, so a single loud frame proves
nothing — this requires several consecutive frames well above the normal
speech threshold (BARGE_IN_RMS_THRESHOLD defaults to 4x VAD_RMS_THRESHOLD).
Takes the same injectable stream shape as mic.record_utterance, so tests feed
it fake frames instead of real audio hardware.
Two detectors, picked by BARGE_IN_MODE:
- **wake** (default) — the interruption has to be the wake word. Every mic
frame goes through the same openWakeWord model the idle listener uses, so
a sneeze, a door, or the TV can't cut Bolt off mid-sentence; only saying
"thunderbolt" does.
- **energy** — the original behaviour: N consecutive frames above
BARGE_IN_RMS_THRESHOLD. Faster to trigger and needs no model inference,
but it fires on any sustained noise. Deliberately dumber than the
utterance VAD in mic.py, since the mic hears the pet's own voice coming
back out of the speakers, so the threshold defaults to 4x the VAD one.
Both take the same injectable stream shape as mic.record_utterance and expose
the same reset()/check() pair, so tests feed them fake frames instead of real
audio hardware and controller.py doesn't care which one it holds.
"""
from __future__ import annotations
from typing import Callable, Optional, Union
import numpy as np
from .. import config
from .mic import AudioStream, rms
from .wake_word import WakeModel, _default_model, hard_reset
def _read_frame(stream: AudioStream, frame_len: int) -> Optional[np.ndarray]:
"""One mono frame, or None if the mic hiccuped or gave us nothing. Never
raises: a bad frame mid-playback should mean "no barge-in this frame",
not a dead reply."""
try:
chunk, _ = stream.read(frame_len)
except Exception:
return None
frame = np.asarray(chunk)
if frame.ndim > 1:
frame = frame[:, 0]
return frame if frame.size else None
class BargeInDetector:
"""Poll-driven: call check() repeatedly while audio plays. Each call
consumes exactly one mic frame (80ms at the default frame length), which
is also what paces the playback loop's polling."""
"""Energy mode. Poll-driven: call check() repeatedly while audio plays.
Each call consumes exactly one mic frame (80ms at the default frame
length), which is also what paces the playback loop's polling."""
def __init__(
self,
@@ -44,19 +70,121 @@ class BargeInDetector:
def check(self) -> bool:
"""True once the user has been loud for long enough to count as an
interruption. Never raises: a mic hiccup mid-playback should not kill
the reply, it should just mean "no barge-in this frame"."""
try:
chunk, _ = self._stream.read(self._frame_len)
except Exception:
return False
frame = np.asarray(chunk)
if frame.ndim > 1:
frame = frame[:, 0]
if frame.size == 0:
interruption."""
frame = _read_frame(self._stream, self._frame_len)
if frame is None:
return False
if rms(frame) >= self._threshold:
self._loud_frames += 1
else:
self._loud_frames = 0 # a single thump/cough shouldn't count
return self._loud_frames >= self._required
class WakeWordBargeIn:
"""Wake-word mode: only "thunderbolt" interrupts.
Same per-frame predict() loop as listen_for_wake_word, just driven by the
playback poll instead of its own read loop. The model instance is shared
with the idle listener by default — the two never run at the same time
(the pipeline is either speaking or listening), and reusing it avoids
loading a second copy of the ONNX graph.
Two wrinkles the energy detector doesn't have:
- The mic hears the pet's own voice, so the model is scoring Bolt's
speech too. That's harmless unless Bolt says its own wake word, which
is why the threshold can be raised independently
(BARGE_IN_WAKE_THRESHOLD) without desensitizing the idle listener.
- reset() has to be a *hard* reset. openwakeword keeps ~10s of audio
history in its preprocessor, so the "thunderbolt" that started this
turn is still in the model's window when playback begins — feed it one
new frame and it fires on the old phrase, cutting the reply off a word
in. Clearing that window is what makes wake-mode barge-in work at all.
"""
def __init__(
self,
stream: AudioStream,
model: Optional[WakeModel] = None,
threshold: Union[float, Callable[[], float], None] = None,
frame_len: int = config.FRAME_LEN,
on_score: Optional[Callable[[float, float], None]] = None,
):
self._stream = stream
self._model = model if model is not None else _default_model
if threshold is None:
threshold = config.BARGE_IN_WAKE_THRESHOLD or config.WAKE_WORD_THRESHOLD
self._resolve_threshold = threshold if callable(threshold) else (lambda: threshold)
self._frame_len = frame_len
self._on_score = on_score
self._frames = 0
self._peak = 0.0
self._last = 0.0
self._last_threshold = 0.0
# Scoring history for the current reply. Without this an interruption is
# indistinguishable from a crash in the logs — you can't tell a genuine
# "thunderbolt" from the model firing on Bolt's own voice, or on the first
# frame (a stale window) versus halfway through (something it heard).
@property
def frames_checked(self) -> int:
return self._frames
@property
def seconds_checked(self) -> float:
return self._frames * self._frame_len / config.SAMPLE_RATE
@property
def peak_score(self) -> float:
return self._peak
@property
def last_score(self) -> float:
return self._last
@property
def last_threshold(self) -> float:
return self._last_threshold
def reset(self) -> None:
hard_reset(self._model) # never raises
self._frames = 0
self._peak = 0.0
self._last = 0.0
def check(self) -> bool:
frame = _read_frame(self._stream, self._frame_len)
if frame is None:
return False
try:
scores = self._model.predict(frame)
except Exception:
return False # same contract as a mic hiccup: no barge-in, no crash
threshold = self._resolve_threshold()
best = max(scores.values()) if scores else 0.0
self._frames += 1
self._last = best
self._peak = max(self._peak, best)
self._last_threshold = threshold
if self._on_score is not None:
self._on_score(best, threshold)
if scores and best >= threshold:
self.reset()
return True
return False
def make_detector(
stream: AudioStream,
mode: str = None,
wake_threshold: Union[float, Callable[[], float], None] = None,
on_score: Optional[Callable[[float, float], None]] = None,
):
"""Build whichever detector BARGE_IN_MODE asks for. An unrecognized mode
falls back to energy rather than raising — a typo in .env shouldn't stop
the pet from starting."""
mode = (config.BARGE_IN_MODE if mode is None else mode).strip().lower()
if mode in ("wake", "wakeword", "wake_word"):
return WakeWordBargeIn(stream, threshold=wake_threshold, on_score=on_score)
return BargeInDetector(stream)
+44 -1
View File
@@ -37,6 +37,42 @@ def rms(frame: np.ndarray) -> float:
return float(np.sqrt(np.mean(frame.astype(np.float64) ** 2)))
def flush(stream, max_seconds: float = 10.0, sample_rate: int = config.SAMPLE_RATE) -> int:
"""Throw away whatever is already sitting in the mic's buffer. Returns the
number of frames dropped.
PortAudio keeps capturing into a ring buffer while nothing is reading it, so
audio recorded during a long blocking stretch is still queued when the next
read happens. That matters exactly once: at the end of a reply the pet is
about to listen for an answer, and the last fraction of a second of its own
TTS is in that buffer. It's above the VAD threshold, so `record_utterance`
treats it as the start of your answer, Deepgram transcribes it, and Bolt is
handed his own sentence as if you had said it. With barge-in on, the
detector was draining the stream during playback and the window is small;
with `BARGE_IN=false` nothing drains it at all.
Only safe where the buffer is known to hold *nothing you said* — never
before a wake-triggered recording, where the rest of "thunderbolt, what
time is it" is legitimately queued and dropping it clips the request.
*max_seconds* bounds a single call so this can't chase a stream that's
filling as fast as it's read. Best-effort: a fake stream in tests has no
`read_available` and this is a no-op, which is the correct behaviour for
one."""
try:
available = int(getattr(stream, "read_available", 0) or 0)
except (TypeError, ValueError):
return 0
if available <= 0:
return 0
frames = min(available, int(max_seconds * sample_rate))
try:
stream.read(frames)
except Exception:
return 0 # a mid-flush device error is the reader's problem, not ours
return frames
def record_utterance(
stream: AudioStream,
should_continue=lambda: True,
@@ -44,6 +80,7 @@ def record_utterance(
silence_end_sec: float = None,
max_utterance_s: float = None,
min_utterance_s: float = None,
grace_s: float = None,
frame_len: int = config.FRAME_LEN,
sample_rate: int = config.SAMPLE_RATE,
) -> Optional[np.ndarray]:
@@ -53,18 +90,24 @@ def record_utterance(
*should_continue* is polled each frame so a caller can cancel recording
(e.g. the pet window was closed) without needing threading primitives
baked into this function.
*grace_s* is how long to wait for speech to *begin* before giving up.
The controller stretches it for follow-up questions, where you're being
asked something and need a moment to think rather than having just said
the wake word on purpose.
"""
rms_threshold = config.RMS_THRESHOLD if rms_threshold is None else rms_threshold
silence_end_sec = config.SILENCE_END_SEC if silence_end_sec is None else silence_end_sec
max_utterance_s = config.MAX_UTTERANCE_S if max_utterance_s is None else max_utterance_s
min_utterance_s = config.MIN_UTTERANCE_S if min_utterance_s is None else min_utterance_s
grace_s = config.GRACE_SECONDS if grace_s is None else grace_s
frames: list[np.ndarray] = []
started = False
silence_frames = 0
silence_limit = int(silence_end_sec * sample_rate / frame_len)
max_frames = int(max_utterance_s * sample_rate / frame_len)
grace_frames = int(4.0 * sample_rate / frame_len) # wait up to 4s for speech to begin
grace_frames = int(grace_s * sample_rate / frame_len) # how long to wait for speech to begin
waited = 0
while should_continue():
+96 -12
View File
@@ -5,11 +5,16 @@ desk_client/bolt_desk.py which shells out because it only targets Linux.
Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech
on macOS, espeak on Linux) if ElevenLabs isn't configured or the request
fails, so the pet can still talk with zero cloud config.
Every entry point takes an optional *voice_id* that overrides
`ELEVENLABS_VOICE_ID` for that call — that's how the server's `speak_as`
reply marker reaches the speakers (see controller._apply_voice). The offline
fallback has no such concept and always sounds like itself.
"""
from __future__ import annotations
from typing import Iterable, Iterator
from typing import Iterable, Iterator, Optional
import numpy as np
import requests
@@ -21,18 +26,40 @@ class TtsError(Exception):
pass
def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
def voice_for(voice_id: Optional[str] = None) -> str:
"""The voice this call should use: an override (server `speak_as`) if
given, else the configured default."""
return (voice_id or "").strip() or config.ELEVENLABS_VOICE_ID
def model_for(text: str, voice_id: Optional[str] = None) -> str:
"""Which ElevenLabs model to synthesize with.
The default (`eleven_flash_v2`) is English-only, and both things that
reach this branch mean the reply probably isn't English: a voice the
server picked mid-conversation is nearly always about a language or an
accent, and non-ASCII text can't be English at all. Rendering either one
through the English model gets you a mangled phonetic reading rather
than a failure, which is worse — so those go through the multilingual
model instead."""
if (voice_id or "").strip() or not text.isascii():
return config.ELEVENLABS_MULTILINGUAL_MODEL_ID
return config.ELEVENLABS_MODEL_ID
def synthesize_pcm(text: str, voice_id: Optional[str] = None) -> tuple[np.ndarray, int]:
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
callers should fall back to speak_offline() rather than treating this
as fatal."""
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
voice = voice_for(voice_id)
if not (config.ELEVENLABS_API_KEY and voice):
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
try:
response = requests.post(
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}",
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
json={"text": text, "model_id": model_for(text, voice_id)},
timeout=60,
)
response.raise_for_status()
@@ -44,20 +71,23 @@ def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
return pcm, config.TTS_SAMPLE_RATE
def stream_pcm(text: str, chunk_bytes: int = 4096) -> Iterator[np.ndarray]:
def stream_pcm(
text: str, chunk_bytes: int = 4096, voice_id: Optional[str] = None
) -> Iterator[np.ndarray]:
"""Same audio as synthesize_pcm(), but yielded as it arrives from
ElevenLabs' /stream endpoint so playback can start on the first chunk
(~300ms) instead of after the whole clip is synthesized. Raises TtsError
before yielding anything if the request itself fails, so callers can fall
back cleanly; a mid-stream failure just ends the generator."""
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
voice = voice_for(voice_id)
if not (config.ELEVENLABS_API_KEY and voice):
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
try:
response = requests.post(
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}/stream",
f"https://api.elevenlabs.io/v1/text-to-speech/{voice}/stream",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
json={"text": text, "model_id": model_for(text, voice_id)},
timeout=60,
stream=True,
)
@@ -83,6 +113,55 @@ def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
yield np.frombuffer(data[:usable], dtype=np.int16)
def synthesize_dialogue(
inputs: list, model_id: Optional[str] = None, stability: Optional[float] = None
) -> tuple[np.ndarray, int]:
"""Multi-voice scene via ElevenLabs Text to Dialogue.
One request, one take: the whole exchange is synthesized together, which
is the point — the model hears the previous line, so reactions and timing
land instead of sounding like separately-rendered clips.
Same PCM-over-`requests` posture as the rest of this module (no SDK, no
`play()` shelling out to ffplay), so playback is the same sounddevice path
everything else uses and barge-in works on it unchanged. There is no
documented streaming variant, and a scene is a short set piece anyway, so
this is whole-clip only.
"""
if not (config.ELEVENLABS_API_KEY and inputs):
raise TtsError("ELEVENLABS_API_KEY not set (or no dialogue lines)")
body: dict = {
"inputs": [
{"text": str(entry.get("text") or ""), "voice_id": str(entry.get("voice_id") or "")}
for entry in inputs
],
"model_id": model_id or config.DIALOGUE_MODEL_ID,
}
if stability is not None:
body["settings"] = {"stability": float(stability)}
try:
response = requests.post(
"https://api.elevenlabs.io/v1/text-to-dialogue",
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
json=body,
timeout=120, # a multi-voice take is slower to render than one line
)
response.raise_for_status()
except Exception as exc:
detail = ""
# The API explains refusals (character limit, unknown voice) in the
# body; surfacing it is what lets Bolt fix the call and retry.
body_text = getattr(getattr(exc, "response", None), "text", "")
if body_text:
detail = f"{body_text[:300]}"
raise TtsError(f"ElevenLabs dialogue request failed: {exc}{detail}") from exc
pcm = np.frombuffer(response.content, dtype=np.int16)
if pcm.size == 0:
raise TtsError("ElevenLabs returned no dialogue audio")
return pcm, config.TTS_SAMPLE_RATE
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None) -> bool:
"""Play a whole clip. Returns True if it finished, False if *should_stop*
(barge-in) cut it short. *should_stop* is polled while audio plays — each
@@ -134,11 +213,12 @@ def speak_offline(text: str) -> None:
engine.runAndWait()
def speak(text: str, on_error=None, should_stop=None) -> bool:
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None) -> bool:
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
exception when ElevenLabs fails (useful for logging) — a fallback still
runs either way. Returns False if barge-in interrupted playback.
*voice_id* overrides the configured voice for this line only.
The text is sanitized first (speech_text.for_speech): server replies are
written for a chat window, and a voice reads markdown/emoji literally
@@ -149,12 +229,16 @@ def speak(text: str, on_error=None, should_stop=None) -> bool:
return True
if config.TTS_STREAMING:
try:
return play_stream(stream_pcm(text), config.TTS_SAMPLE_RATE, should_stop=should_stop)
return play_stream(
stream_pcm(text, voice_id=voice_id),
config.TTS_SAMPLE_RATE,
should_stop=should_stop,
)
except TtsError as exc:
if on_error is not None:
on_error(exc)
try:
pcm, sample_rate = synthesize_pcm(text)
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
return play_pcm(pcm, sample_rate, should_stop=should_stop)
except TtsError as exc:
if on_error is not None:
+77 -3
View File
@@ -64,6 +64,65 @@ class NearMissLog:
self._peak = 0.0
# Where the cached blank state is stashed — on the preprocessor itself
# rather than in a dict keyed by id(), which CPython reuses after garbage
# collection and would hand one model another's buffers.
_BLANK_STATE_ATTR = "_bolt_blank_state"
def _blank_state(preprocessor) -> Optional[tuple]:
"""(feature_buffer, melspectrogram_buffer) as they are on a freshly
constructed model — i.e. "having heard nothing". Computing it costs an
ONNX pass over 10s of silence, so it's cached on the preprocessor and
copied from thereafter. Returns None if openwakeword's internals don't
look the way we expect, in which case callers leave the state alone
rather than corrupting it."""
cached = getattr(preprocessor, _BLANK_STATE_ATTR, None)
if cached is not None:
return cached
try:
# Same call the AudioFeatures constructor uses to prime the buffer.
state = (preprocessor._get_embeddings(np.zeros(160000).astype(np.int16)), np.ones((76, 32)))
setattr(preprocessor, _BLANK_STATE_ATTR, state)
except Exception:
return None
return state
def hard_reset(model) -> None:
"""Make the model forget the audio it has already heard — not just its
predictions.
openwakeword's ``Model.reset()`` clears the *prediction* buffer only.
The rolling audio window the classifier actually scores lives in
``model.preprocessor`` (raw_data_buffer / melspectrogram_buffer /
feature_buffer, ~10s of history) and has no reset method of its own. So
after a detection the wake word is still sitting in that window, and the
next frame fed to the model re-fires on it — which is exactly what made
the pet interrupt itself a word into every reply: the "thunderbolt" that
started the turn was still in the buffer when barge-in resumed feeding it.
Never raises. A model whose internals don't match (a fake in tests, a
future openwakeword release) just gets the plain reset()."""
try:
model.reset()
except Exception:
pass
preprocessor = getattr(model, "preprocessor", None)
if preprocessor is None:
return
blank = _blank_state(preprocessor)
try:
if blank is not None:
features, melspectrogram = blank
preprocessor.feature_buffer = features.copy()
preprocessor.melspectrogram_buffer = melspectrogram.copy()
preprocessor.raw_data_buffer.clear()
preprocessor.accumulated_samples = 0
except Exception:
pass
def _construct_model(model_cls, model_path: str):
"""openwakeword's Model() constructor keyword has drifted across
releases (wakeword_models -> wakeword_model_paths) and some builds
@@ -112,6 +171,17 @@ class _OpenWakeWordModel:
self._model = _construct_model(Model, config.WAKE_MODEL_PATH)
return self._model
@property
def preprocessor(self):
"""Proxy the wrapped model's audio-feature buffers.
Without this, hard_reset() sees a wrapper with no `preprocessor` and
silently degrades to openwakeword's shallow reset() — which leaves the
previous detection sitting in the audio window, i.e. exactly the bug
hard_reset exists to fix. Returns None before the model is loaded, so
a reset that happens first is a no-op rather than a load."""
return getattr(self._model, "preprocessor", None)
def predict(self, frame: np.ndarray) -> dict:
return self._ensure_model().predict(frame)
@@ -136,8 +206,9 @@ def listen_for_wake_word(
Feeds every frame to *model* (the thunderbolt openWakeWord model by
default) and treats any class score >= *threshold* as a detection,
resetting the model's internal state afterward so the next call starts
clean — same pattern as desk_client/bolt_desk.py's main loop.
clearing the model's internal state (audio window included, see
hard_reset) afterward so the next call starts clean — same pattern as
desk_client/bolt_desk.py's main loop.
*threshold* may be a number or a zero-argument callable. The callable
form exists because this function blocks for minutes at a time: the
@@ -180,6 +251,9 @@ def listen_for_wake_word(
on_score(best, current_threshold)
if scores and best >= current_threshold:
model.reset()
# hard_reset, not reset: the phrase has to leave the model's audio
# window too, or the very next frame we feed it re-fires on the
# same "thunderbolt" (see hard_reset's docstring).
hard_reset(model)
return True
return False
+153 -5
View File
@@ -66,9 +66,38 @@ DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3")
ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "")
ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "")
ELEVENLABS_MODEL_ID = os.environ.get("ELEVENLABS_MODEL_ID", "eleven_flash_v2")
# eleven_flash_v2 is English-only, and the two cases that swap the voice
# (server-picked `speak_as`, or a reply with non-ASCII in it) are usually
# exactly the cases where the reply isn't English — see tts.model_for().
ELEVENLABS_MULTILINGUAL_MODEL_ID = os.environ.get(
"ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5"
)
# ElevenLabs PCM output formats are named pcm_<sample_rate>.
TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "24000"))
# Does a voice the server picks (its speak_as marker — "talk like a pirate",
# "say that in Japanese") stay on for later replies, or last one reply only?
# Sticky by default: the server tags a single reply and does *not* keep the
# voice id in its history, so a one-reply-only voice can't be re-used when
# you say "keep talking like that" — it would have to search for a voice
# again. Reset it from the tray ("Use default voice") or by restarting.
VOICE_STICKY = os.environ.get("VOICE_STICKY", "true").lower() in ("1", "true", "yes", "on")
# ── multi-voice dialogue (ElevenLabs Text to Dialogue) ──────────────────────
# Lets Bolt play a short scene in several voices with delivery tags the v3
# model acts on ("[cheerfully] Hello"), instead of one voice reading a line.
# Driven by the server through the `dialoguectl` relayed command — see
# dialogue.py. Costs a separate (slower, whole-clip) request per scene, so
# it's a set piece, not the normal reply path.
#
# DIALOGUE_VOICES names the cast: "narrator:9BWtsMINqrJLrRacOk9x,villain:IKne3meq5aSn9XLyUdCD".
# The name "self" always resolves to the voice the pet is currently using,
# including one the server picked with speak_as.
DIALOGUE = os.environ.get("DIALOGUE", "true").lower() in ("1", "true", "yes", "on")
DIALOGUE_MODEL_ID = os.environ.get("DIALOGUE_MODEL_ID", "eleven_v3")
DIALOGUE_VOICES = os.environ.get("DIALOGUE_VOICES", "")
# ── mic / VAD (same tuning knobs as bolt_desk.py) ───────────────────────────
MIC_DEVICE = os.environ.get("MIC_DEVICE", "") or None # sounddevice name/index
@@ -78,19 +107,74 @@ RMS_THRESHOLD = int(os.environ.get("VAD_RMS_THRESHOLD", "300"))
SILENCE_END_SEC = float(os.environ.get("VAD_SILENCE_END_SEC", "1.2"))
MAX_UTTERANCE_S = float(os.environ.get("VAD_MAX_UTTERANCE_SECONDS", "15"))
MIN_UTTERANCE_S = float(os.environ.get("VAD_MIN_UTTERANCE_SECONDS", "0.4"))
# How long to wait for you to *start* talking before giving up on a turn.
GRACE_SECONDS = float(os.environ.get("VAD_GRACE_SECONDS", "4"))
# ── follow-up listening ─────────────────────────────────────────────────────
# When a reply ends on a question, the pet keeps listening for the answer
# instead of dropping back to idle and making you say the wake word again.
# The grace period is longer than a normal turn's because you were asked
# something and may need a beat to think. FOLLOW_UP_MAX_TURNS caps how many
# question-and-answer rounds can chain without you re-triggering it — a stop
# on runaway loops if the server ends every reply with a question and the mic
# keeps feeding it noise. 0 means no cap.
FOLLOW_UP_LISTEN = os.environ.get("FOLLOW_UP_LISTEN", "true").lower() in ("1", "true", "yes", "on")
FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "10"))
FOLLOW_UP_GRACE_SECONDS = float(os.environ.get("FOLLOW_UP_GRACE_SECONDS", "7"))
# ── local intents ───────────────────────────────────────────────────────────
# A short, closed list of utterances the pet answers itself instead of paying a
# server round trip for: "stop", "come here", "go to sleep", "say that again",
# "use your normal voice". Matched whole and exact (see intents.py), never
# during a follow-up turn, so a real request is never swallowed. Turn it off to
# route absolutely everything through Bolt.
LOCAL_INTENTS = os.environ.get("LOCAL_INTENTS", "true").lower() in ("1", "true", "yes", "on")
COMMAND_TIMEOUT_SECONDS = int(os.environ.get("COMMAND_TIMEOUT_SECONDS", "30"))
# ── sudo password prompts ───────────────────────────────────────────────────
# The pet has no terminal, so a relayed `sudo` would block on a tty nobody is
# watching. With this on, bare `sudo` is rewritten to `sudo -A` and the
# password is collected in a desktop dialog (a real askpass binary if one is
# installed, otherwise a generated zenity/kdialog wrapper). Turn it off and
# sudo commands simply fail, which is the safer default if you'd rather Bolt
# never be able to ask for root at all.
SUDO_ASKPASS_PROMPT = os.environ.get("SUDO_ASKPASS_PROMPT", "true").lower() in ("1", "true", "yes", "on")
SUDO_ASKPASS_HELPER = os.environ.get("SUDO_ASKPASS_HELPER", "") # blank = auto-detect
# Longer than COMMAND_TIMEOUT_SECONDS because a person has to notice the
# dialog, read it, and type — 30s is nowhere near enough for that.
SUDO_COMMAND_TIMEOUT_SECONDS = int(os.environ.get("SUDO_COMMAND_TIMEOUT_SECONDS", "180"))
HEARTBEAT_INTERVAL_SECONDS = float(os.environ.get("HEARTBEAT_INTERVAL_SECONDS", "60"))
# ── barge-in (interrupt playback by talking over it) ────────────────────────
# The mic stays live while the pet talks; sustained loud frames cut playback
# short. The threshold is deliberately well above VAD_RMS_THRESHOLD because
# the mic also hears the pet's own voice through the speakers — raise it
# further (or set BARGE_IN=false) if playback keeps interrupting itself.
# ── self-restart ────────────────────────────────────────────────────────────
# `petctl self_restart` lets Bolt restart the pet after editing its code, so
# he can see his own change running instead of waiting for someone to restart
# it by hand. The code is import-checked in a subprocess first, and the reason
# is carried across the restart so the new process can report back — see
# self_restart.py. SELF_RESTART_MAX/_WINDOW_SECONDS bound the crash-loop case.
SELF_RESTART = os.environ.get("SELF_RESTART", "true").lower() in ("1", "true", "yes", "on")
# ── barge-in (interrupt playback while the pet is talking) ──────────────────
# The mic stays live while the pet talks. BARGE_IN_MODE decides what counts
# as an interruption:
# wake — only the wake word cuts playback (default). Immune to coughs,
# doors, and the TV, at the cost of ~a word of extra latency.
# energy — any sustained noise above BARGE_IN_RMS_THRESHOLD does. Faster,
# but interrupts on background noise. That threshold is well above
# VAD_RMS_THRESHOLD because the mic also hears the pet's own voice
# through the speakers — raise it further if playback keeps
# interrupting itself.
# Set BARGE_IN=false to make playback uninterruptible either way.
BARGE_IN = os.environ.get("BARGE_IN", "true").lower() in ("1", "true", "yes", "on")
BARGE_IN_MODE = os.environ.get("BARGE_IN_MODE", "wake")
BARGE_IN_RMS_THRESHOLD = int(os.environ.get("BARGE_IN_RMS_THRESHOLD", str(RMS_THRESHOLD * 4)))
BARGE_IN_FRAMES = int(os.environ.get("BARGE_IN_FRAMES", "4")) # consecutive loud frames (80ms each)
# Wake-mode sensitivity. Blank means "track the live WAKE_WORD_THRESHOLD from
# the tray tuner"; set a number to make interrupting deliberately harder than
# waking the pet from idle (useful if Bolt's own voice trips the model).
BARGE_IN_WAKE_THRESHOLD = float(os.environ.get("BARGE_IN_WAKE_THRESHOLD") or 0) or None
# ── streaming TTS ───────────────────────────────────────────────────────────
# ElevenLabs' /stream endpoint + chunked playback: the pet starts talking
@@ -105,6 +189,30 @@ TTS_STREAMING = os.environ.get("TTS_STREAMING", "true").lower() in ("1", "true",
SCREEN_CONTEXT = os.environ.get("SCREEN_CONTEXT", "true").lower() in ("1", "true", "yes", "on")
# ── monitors ────────────────────────────────────────────────────────────────
# A one-line note about the screen layout (how many, their sizes, which one
# the pet is standing on) rides along with each utterance, so Bolt can decide
# to `petctl jump` somewhere without asking you what you've got plugged in.
# Cheap — the list comes from the UI, nothing is probed per turn.
MONITOR_CONTEXT = os.environ.get("MONITOR_CONTEXT", "true").lower() in (
"1", "true", "yes", "on"
)
# ── screen text (OCR) ───────────────────────────────────────────────────────
# Lets Bolt actually read a monitor, via `petctl read`. Pull-only: nothing is
# captured unless the server asks for it, and every read is logged. Needs the
# optional capture/OCR extras — see the comments in requirements.txt.
#
# This widens what can leave the machine more than any other switch here: the
# recognised text of a whole screen goes to the server. It is *not* a new
# capability (the shell relay could already run a screenshot tool and OCR it),
# but it is a much easier one to use by accident. Set SCREEN_TEXT=false to
# take it away entirely.
SCREEN_TEXT = os.environ.get("SCREEN_TEXT", "true").lower() in ("1", "true", "yes", "on")
SCREEN_TEXT_MAX_CHARS = int(os.environ.get("SCREEN_TEXT_MAX_CHARS", "4000"))
# ── quiet hours / do-not-disturb ────────────────────────────────────────────
# Comma-separated HH:MM-HH:MM ranges (wrapping midnight is fine). While
# napping the pet dims, stops wandering, and makes no proactive noise —
@@ -121,6 +229,25 @@ NOTIFICATION_BRIDGE = os.environ.get("NOTIFICATION_BRIDGE", "false").lower() in
# Regex matched against "<app>: <summary> <body>"; empty means "everything".
NOTIFICATION_FILTER = os.environ.get("NOTIFICATION_FILTER", "")
NOTIFICATION_MIN_INTERVAL_SECONDS = float(os.environ.get("NOTIFICATION_MIN_INTERVAL_SECONDS", "60"))
# Notifications arrive on the watcher thread and are forwarded from the
# heartbeat, which doesn't run while the pet is napping — so they queue. Both
# limits exist to stop an overnight backlog turning into a burst of round trips
# and a monologue at 8am: the queue is bounded (oldest dropped first) and
# anything staler than the age limit is discarded at drain time, because
# "Firefox finished downloading" is not news nine hours later.
NOTIFICATION_QUEUE_LIMIT = int(os.environ.get("NOTIFICATION_QUEUE_LIMIT", "20"))
NOTIFICATION_MAX_AGE_SECONDS = float(os.environ.get("NOTIFICATION_MAX_AGE_SECONDS", "900"))
# ── file delivery ────────────────────────────────────────────────────────
# The server's deliver_files tool (ai/desk_api.py in the main tmn-api repo)
# queues workspace files on this session — e.g. "send me that report" — for
# the client to fetch via GET /desk/files. Downloading a file dequeues it
# server-side, so each one lands here exactly once.
RECEIVE_FILES = os.environ.get("RECEIVE_FILES", "true").lower() in ("1", "true", "yes", "on")
DELIVERED_FILES_DIR = Path(
os.environ.get("DELIVERED_FILES_DIR") or str(Path.home() / "Downloads" / "Bolt")
).expanduser()
# ── conversation history ────────────────────────────────────────────────────
@@ -140,6 +267,27 @@ PUSH_TO_TALK_HOTKEY = os.environ.get("PUSH_TO_TALK_HOTKEY", "ctrl+alt+space")
WAKE_NEAR_MISS_MARGIN = float(os.environ.get("WAKE_NEAR_MISS_MARGIN", "0.2"))
WAKE_NEAR_MISS_LIMIT = int(os.environ.get("WAKE_NEAR_MISS_LIMIT", "40"))
# ── auto-update ─────────────────────────────────────────────────────────────
# Watches the Gitea releases API for a tag newer than bolt_pet.__version__,
# then `git checkout`s it in place and restarts (see updater.py). The install
# has to be a git clone with a clean working tree — a dirty tree is skipped
# rather than stashed, so local edits are never thrown away. Any failure
# after checkout rolls back to the ref that was checked out before.
AUTO_UPDATE = os.environ.get("AUTO_UPDATE", "true").lower() in ("1", "true", "yes", "on")
UPDATE_REPO_API = os.environ.get(
"UPDATE_REPO_API",
"https://git.themajesticnetwork.com/api/v1/repos/TheMajesticNetwork/Bolt-Pet",
).rstrip("/")
UPDATE_CHECK_INTERVAL_SECONDS = float(os.environ.get("UPDATE_CHECK_INTERVAL_SECONDS", "3600"))
UPDATE_GIT_REMOTE = os.environ.get("UPDATE_GIT_REMOTE", "origin")
# Only needed if the repo is private — releases on a public repo read fine
# anonymously. A Gitea access token with read:repository.
UPDATE_TOKEN = os.environ.get("UPDATE_TOKEN", "")
# Reinstall requirements.txt when an update changes it. Off means a release
# that adds a dependency will roll straight back on the import smoke test.
UPDATE_INSTALL_DEPS = os.environ.get("UPDATE_INSTALL_DEPS", "true").lower() in ("1", "true", "yes", "on")
SAMPLE_RATE = 16000 # mic capture / STT rate
FRAME_LEN = 1280 # 80ms @ 16kHz — matches bolt_desk.py's chunking
+650 -22
View File
@@ -15,11 +15,18 @@ from __future__ import annotations
import threading
import time
from collections import deque
from typing import Optional
from PySide6.QtCore import QObject, Signal
from . import config, history as history_mod, notifications, pet_actions, quiet, screen_context, server_client, speech_text
from . import (
config, dialogue as dialogue_mod, file_delivery, file_ops,
history as history_mod, intents as intents_mod, monitors as monitors_mod,
notifications, pet_actions, quiet, screen_context, screen_text,
self_restart, server_client, speech_text, updater,
)
from . import __version__
from .audio import barge_in, mic, stt, tts, wake_word
from .state import PetState, PetStateMachine
@@ -34,6 +41,8 @@ class PetController(QObject):
log = Signal(str)
action = Signal(dict) # parsed petctl action for the UI to perform
napping = Signal(bool) # quiet hours / fullscreen do-not-disturb
voice_changed = Signal(str) # name of the server-picked voice ("" = default)
restart_requested = Signal(str) # version we just updated to
finished = Signal()
def __init__(self):
@@ -49,21 +58,55 @@ class PetController(QObject):
# window. Append-only from this thread; the UI only ever snapshots it.
self.history = history_mod.ConversationHistory(limit=config.HISTORY_LIMIT)
# The screen layout, as published by the UI (see set_monitors). Held
# here rather than probed, so "monitor 2" means the same thing to the
# controller and to the window that has to jump there — see
# monitors.py for why that matters.
self._monitors: list[monitors_mod.Monitor] = []
self._pet_monitor: Optional[int] = None
# Wake-word sensitivity is live-tunable (tray tuner), so it's read
# through a callable on every frame rather than captured per listen.
self._wake_threshold = config.WAKE_WORD_THRESHOLD
self._near_misses = wake_word.NearMissLog()
# The voice the server last picked for us with `speak_as` ("" = the
# configured default). Held here rather than passed straight through
# to one tts.speak() call because it's sticky by default — see
# _apply_voice for why.
self._voice_id = ""
self._voice_name = ""
self._barge_in: Optional[barge_in.BargeInDetector] = None
self._napping = False
self._nap_forced: Optional[bool] = None # petctl nap on/off overrides the schedule
self._last_nap_check = 0.0
# When a reply ends on a question the pet keeps listening for the
# answer. _follow_ups counts how many have chained without you
# re-triggering, so a server that ends every reply with "?" can't
# loop forever off mic noise.
self._pending_follow_up = False
self._follow_ups = 0
self._last_update_check = 0.0
self._update_pending = False # applied on disk, waiting for the restart
# Armed by `petctl self_restart`, fired after the turn it was asked in
# (see _arm_self_restart for why it can't happen inline).
self._restart_context = None
self._notification_watcher: Optional[notifications.NotificationWatcher] = None
self._notification_gate = notifications.NotificationGate(
config.NOTIFICATION_FILTER, config.NOTIFICATION_MIN_INTERVAL_SECONDS
)
self._pending_notifications: list[notifications.Notification] = []
# Bounded, and stamped on arrival: the drain only runs from the
# heartbeat, which doesn't run while napping, so this fills up
# overnight. maxlen drops the oldest rather than growing without limit,
# and the stamp lets the drain discard a backlog nobody wants read out
# at 8am (see _drain_notifications).
self._pending_notifications: deque[tuple[float, notifications.Notification]] = deque(
maxlen=max(1, config.NOTIFICATION_QUEUE_LIMIT)
)
self._notification_lock = threading.Lock()
# ── external controls (safe to call from the Qt/UI thread) ─────────
@@ -95,6 +138,20 @@ class PetController(QObject):
def reset_wake_stats(self) -> None:
self._near_misses.clear()
def current_voice(self) -> str:
"""Name (or id) of the server-picked voice in use, "" for the default."""
return self._voice_name or self._voice_id
def reset_voice(self) -> None:
"""Drop a server-picked voice and go back to Bolt's own. The tray's
way out of a voice you didn't want to keep — the server has no way to
ask for the default back, since it never learns what it is."""
if not self._voice_id:
return
self._voice_id = self._voice_name = ""
self.log.emit("Voice: back to the default.")
self.voice_changed.emit("")
def stop(self) -> None:
self._running = False
self._talk_now.set() # wake up anything blocked waiting on it
@@ -124,8 +181,27 @@ class PetController(QObject):
return
if config.BARGE_IN:
self._barge_in = barge_in.BargeInDetector(self._stream)
# In wake mode the detector shares the idle listener's model and
# its live threshold, so the tray tuner's slider applies to
# interrupting as well as waking (unless BARGE_IN_WAKE_THRESHOLD
# pins it to a fixed, stricter number).
self._barge_in = barge_in.make_detector(
self._stream,
wake_threshold=(
config.BARGE_IN_WAKE_THRESHOLD
if config.BARGE_IN_WAKE_THRESHOLD is not None
else self.wake_threshold
),
)
self.log.emit(f"Barge-in: {config.BARGE_IN_MODE} mode.")
# Everything past here is in try/finally because `finished` is what
# ui/app.py waits on to quit the QThread and to run a pending
# os.execv. An exception escaping _loop used to skip it, leaving the
# thread wedged with the mic still open and no restart — so the failure
# mode of any bug below was "the pet goes deaf and the tray won't quit"
# rather than "one turn failed".
try:
with self._stream:
try:
health = server_client.check_health()
@@ -133,11 +209,34 @@ class PetController(QObject):
except Exception as exc:
self.log.emit(f"Server not reachable yet ({exc}) — will keep trying per-request.")
self._start_notification_bridge()
self._guarded(self._report_self_restart, "restart report")
self._loop()
except Exception as exc:
self.log.emit(f"Pipeline stopped unexpectedly: {exc!r}")
finally:
if self._notification_watcher is not None:
self._notification_watcher.stop()
self.finished.emit()
def _guarded(self, work, label: str) -> bool:
"""Run *work*, absorbing anything it raises.
The pipeline is one thread driving a state machine that raises on an
illegal transition (deliberately — see state.py), plus a dozen
best-effort subsystems that shell out, hit the network, or touch the
filesystem. Any one of them raising something unforeseen used to end the
whole session. Here, it costs a log line and a forced return to IDLE,
which is the only state it's always safe to resume from.
Returns True if *work* completed without raising."""
try:
work()
return True
except Exception as exc:
self.log.emit(f"Recovered from a {label} failure: {exc!r}")
self._state.force(PetState.IDLE)
return False
def _loop(self) -> None:
while self._running:
if self._muted:
@@ -152,7 +251,7 @@ class PetController(QObject):
if not self._running:
return
continue
self._handle_conversation_turn()
self._guarded(self._handle_conversation_turn, "conversation turn")
def _wait_for_wake_or_click(self) -> bool:
"""True once either the wake phrase was heard or a click-to-talk
@@ -164,7 +263,12 @@ class PetController(QObject):
self._stream,
should_continue=should_continue,
threshold=self.wake_threshold, # callable: the tuner slider is live
on_tick=self._maybe_heartbeat,
# Guarded: on_tick is the one place control returns to us during a
# listen that can block for minutes, and everything it drives
# (update check, nap probe, notification forwarding) touches the
# network or shells out. Unguarded, any of them raising would unwind
# the listen loop and end the session.
on_tick=lambda: self._guarded(self._maybe_heartbeat, "heartbeat"),
on_score=self._observe_wake_score,
)
if not self._running:
@@ -181,9 +285,21 @@ class PetController(QObject):
self._near_misses.observe(score, threshold, time.time())
def _handle_conversation_turn(self) -> None:
# A turn you started yourself ends any follow-up chain in progress.
following_up, self._pending_follow_up = self._pending_follow_up, False
if not following_up:
self._follow_ups = 0
self._state.transition(PetState.LISTENING)
pcm = mic.record_utterance(self._stream, should_continue=self._should_continue)
pcm = mic.record_utterance(
self._stream,
should_continue=self._should_continue,
# Answering a question deserves longer than saying the wake word
# on purpose does — you were just asked something.
grace_s=config.FOLLOW_UP_GRACE_SECONDS if following_up else None,
)
if pcm is None:
self._follow_ups = 0 # silence ends the chain
self._state.transition(PetState.IDLE)
return
@@ -201,11 +317,18 @@ class PetController(QObject):
self.log.emit(f"You: {text}")
self.history.add(history_mod.USER, text, time.time())
# "stop", "come here", "say that again" — answered here, without the
# round trip. Never on a follow-up turn: Bolt asked you something and
# the answer is his, even if it happens to look like a body command.
if not following_up and self._handle_local_intent(text):
self._state.transition(PetState.IDLE)
return
try:
# What's focused right now rides along, so "what's this error?"
# has a referent without you having to describe the window.
reply = server_client.converse(
screen_context.context_for(text), on_command=self._handle_command
self._with_context(text), on_command=self._handle_command
)
except server_client.ServerError as exc:
self.log.emit(f"Server error: {exc}")
@@ -213,33 +336,366 @@ class PetController(QObject):
self._state.transition(PetState.IDLE)
return
self._speak(reply)
self._check_deliveries()
self._apply_voice(reply)
self._speak(reply.text)
self._state.transition(PetState.IDLE)
self._maybe_self_restart()
def _handle_local_intent(self, text: str) -> bool:
"""Answer *text* locally if it's one of the closed set of body commands
in intents.py. Returns True if it was handled (no server call).
The effects live here rather than in intents.py for the same reason
pet_actions splits parse from describe: recognising the phrase is pure
and testable, doing the thing needs the controller's state, the tray's
nap override and a Qt signal to the window."""
if not config.LOCAL_INTENTS:
return False
intent = intents_mod.recognize(text)
if intent is None:
return False
self.log.emit(f"Local intent: {intent.name} (answered without the server)")
if intent.name == "stop":
# Nothing to say and nothing to do: silence is the acknowledgement.
# Also ends any follow-up chain — "never mind" means the
# conversation is over, not that we should keep the mic open.
self._follow_ups = 0
self._pending_follow_up = False
self._talk_now.clear()
return True
if intent.name == "repeat":
last = self.history.last(history_mod.PET)
if last is None:
self._speak("I haven't said anything yet.")
else:
# remember=False: replaying a line isn't a new turn. Appending it
# would make "say that again" twice over read back as a
# conversation where Bolt volunteered the same thing three times.
self._speak(last.text, remember=False)
return True
if intent.name == "voice_reset":
had_voice = bool(self._voice_id)
self.reset_voice()
self._speak(intent.speak if had_voice else "That is my normal voice.")
return True
action = intent.action
if action is not None:
if action.get("action") == "nap":
# Through set_napping, not just the signal, so a spoken "go to
# sleep" overrides the quiet-hours schedule exactly like the
# tray's Nap entry and `petctl nap` do — otherwise the next
# schedule check would undo it within ten seconds.
self.set_napping(bool(action["enabled"]))
self.action.emit(dict(action))
if intent.speak:
self._speak(intent.speak)
return True
def _with_context(self, text: str) -> str:
"""Everything the server gets alongside what you actually said: the
focused window title, and a one-line note about the screen layout so
Bolt knows how many monitors there are and where he's standing
without having to ask. Only the *layout* rides along for free — the
text on those screens costs an OCR pass, so it stays behind
`petctl read`."""
text = screen_context.context_for(text)
if config.MONITOR_CONTEXT:
text = monitors_mod.annotate(text, self._monitors, self._pet_monitor)
return text
# ── screen layout, published by the UI ───────────────────────────────
def set_monitors(self, monitors: list) -> None:
"""Slot: the window telling us what screens exist (queued signal)."""
self._monitors = list(monitors)
self.log.emit(
"Screens: " + (monitors_mod.summary(self._monitors) or "none reported")
)
def set_pet_monitor(self, index: int) -> None:
"""Slot: the window telling us which screen the pet is standing on."""
self._pet_monitor = int(index)
def _handle_command(self, command: str) -> str:
"""Server-relayed command. `petctl ...` drives the pet's body and
never reaches a shell; everything else is a real command, exactly as
before (see the security notes in the README)."""
"""Server-relayed command, with the guarantee the relay depends on: this
always returns a string.
The server is blocked on `/desk/tool_result` while this runs. If it
raises instead of answering, the relay never posts, the turn dies
mid-flight, and the server sits out its own timeout on a conversation it
can't finish — the worst available failure mode, because it's silent on
both ends. Handing the exception back as command output instead means
Bolt can read what went wrong and say so, or try something else, inside
the same turn."""
try:
return self._dispatch_command(command)
except Exception as exc:
self.log.emit(f"Command handler failed: {exc!r}")
return f"[error] the pet couldn't run that: {exc}"
def _dispatch_command(self, command: str) -> str:
"""`petctl ...` drives the pet's body, `dialoguectl ...` plays a scene
and `filectl ...` does local file read/write/edit — none of them ever
reach a shell; everything else is a real command, exactly as before (see
the security notes in the README)."""
try:
action = pet_actions.parse(command)
except pet_actions.ActionError as exc:
self.log.emit(f"petctl: {exc}")
return f"[pet] {exc}"
if action is None:
return server_client.run_local_command(command)
if action is not None:
self.log.emit(f"Pet action: {action}")
if action["action"] == "nap":
# Queries answer from here rather than from pet_actions.describe():
# their output *is* the useful part, and it's what the server reads
# back off the tool-result relay.
kind = action["action"]
if kind == "monitors":
return monitors_mod.describe(self._monitors, self._pet_monitor)
if kind == "self_restart":
return self._arm_self_restart(action.get("reason") or "")
if kind == "voice":
# Answered here, not by describe(): the UI has no part in it,
# and the server needs to hear whether there was anything to
# drop — it can't see which voice we're using.
previous = self.current_voice()
self.reset_voice()
return (
f"[pet] back to your own voice (was {previous})" if previous
else "[pet] already using your own voice"
)
if kind == "read":
return self._read_screen(action["target"])
if kind == "jump":
try:
target = monitors_mod.resolve(
self._monitors, action["target"], self._pet_monitor
)
except ValueError as exc:
self.log.emit(f"petctl jump: {exc}")
return f"[pet] {exc}"
# Hand the window a resolved index, so it can't re-resolve the
# spec against a different screen ordering.
self.action.emit({"action": "jump", "monitor": target.index})
return f"[pet] jumped to monitor {target.label}"
if kind == "nap":
self.set_napping(bool(action["enabled"]))
self.action.emit(action)
return pet_actions.describe(action)
def _speak(self, text: str) -> None:
try:
scene = dialogue_mod.parse(command)
except dialogue_mod.DialogueError as exc:
self.log.emit(f"dialoguectl: {exc}")
return f"[dialogue] {exc}"
if scene is not None:
return self._play_dialogue(scene)
try:
file_action = file_ops.parse(command)
except file_ops.FileOpError as exc:
self.log.emit(f"filectl: {exc}")
return f"[filectl] {exc}"
if file_action is not None:
self.log.emit(file_ops.describe(file_action))
try:
return file_ops.execute(file_action)
except file_ops.FileOpError as exc:
self.log.emit(f"filectl: {exc}")
return f"[filectl] {exc}"
return server_client.run_local_command(command)
def _read_screen(self, target: str) -> str:
"""`petctl read` — OCR a screen and hand the text back to the server."""
if not config.SCREEN_TEXT:
return "[pet] screen reading is disabled (set SCREEN_TEXT=true in .env)"
if not self._monitors:
return "[pet] no monitor information available"
limit = config.SCREEN_TEXT_MAX_CHARS
if target in ("all", "everything", "*"):
self.log.emit(f"Reading all {len(self._monitors)} screens…")
return screen_text.read_monitors(self._monitors, limit)
if target in ("here", "", "this", "current"):
index = self._pet_monitor if self._pet_monitor is not None else 0
monitor = self._monitors[min(index, len(self._monitors) - 1)]
else:
try:
monitor = monitors_mod.resolve(
self._monitors, target, self._pet_monitor
)
except ValueError as exc:
return f"[pet] {exc}"
self.log.emit(f"Reading monitor {monitor.number} ({monitor.name})…")
return screen_text.read_monitor(monitor, limit)
def _arm_self_restart(self, reason: str) -> str:
"""`petctl self_restart` — check the code, then arm a restart.
Nothing restarts here. The tool result has to get back up the relay
before this process can die (otherwise the server waits out its
timeout on a turn that will never finish), so the restart is armed and
`_maybe_self_restart` fires it once the turn has been spoken. The
preflight import runs *now*, in this turn, so a syntax error Bolt just
introduced comes back as something he can read and fix rather than as
a pet that never comes back."""
if not config.SELF_RESTART:
return "[pet] self-restart is disabled on this device (SELF_RESTART=false)"
if self._restart_context is not None:
return "[pet] a restart is already armed for the end of this turn"
try:
self_restart.check_loop_guard(self_restart.load())
self.log.emit("Self-restart requested — checking the code imports first…")
self_restart.preflight()
except self_restart.RestartError as exc:
self.log.emit(f"Self-restart refused: {exc}")
return f"[pet] restart refused — {exc}"
recent = [entry.text[:120] for entry in self.history.entries()[-4:]]
self._restart_context = self_restart.arm(
reason or "no reason given",
verify=reason,
version=__version__,
session=config.SESSION_ID,
recent=recent,
)
self.log.emit("Self-restart armed; it happens after this turn.")
return (
"[pet] code imports cleanly; restarting as soon as this turn finishes. "
"I'll come back and tell you what version I'm on and what I found — "
"wrap up your reply now, the next thing you hear from me is the report."
)
def _maybe_self_restart(self) -> bool:
"""Fire an armed restart, once the turn is over and the reply spoken.
Returns True if a restart was requested, so the caller can stop
driving the pipeline — the process is on its way out."""
if self._restart_context is None:
return False
self._update_pending = True # same latch the updater uses: no double restart
self.log.emit("Restarting now.")
self._state.force(PetState.IDLE)
self.restart_requested.emit(f"self-restart: {self._restart_context.reason[:60]}")
return True
def _report_self_restart(self) -> None:
"""On the way up: tell the server we're back, and why we left.
Runs once, before the listen loop starts, and only when a context file
was left behind. The report goes through the ordinary conversation
path, so Bolt's answer is spoken out loud like any other turn — which
is what makes "restart and check the sprites load" finish as a
sentence instead of a silence."""
context = self_restart.load()
if context is None:
return
self_restart.clear()
message = self_restart.report(context, version=__version__)
self.log.emit(f"Back from a self-restart ({context.reason[:80]}).")
self.history.add(history_mod.SYSTEM, message, time.time())
try:
reply = server_client.converse(message, on_command=self._handle_command)
except server_client.ServerError as exc:
# The restart still worked; only the report failed. Say so locally
# rather than pretending nothing happened.
self.log.emit(f"Couldn't report the restart to the server: {exc}")
return
self._apply_voice(reply)
if reply.text.strip() and not self._napping:
self._speak(reply.text)
self._state.force(PetState.IDLE)
def _play_dialogue(self, scene: dict) -> str:
"""Play a `dialoguectl` scene and report back up the relay.
This runs *mid-turn* (the server is still waiting on the tool result),
so the pet has to look like it's talking and then go back to waiting —
hence the TALKING → THINKING leg rather than the usual return to IDLE.
Everything a normal reply gets, a scene gets too: the bubble, the
transcript, and barge-in, so a long scene can be talked over exactly
like a long answer."""
if not config.DIALOGUE:
return "[dialogue] disabled on this device (DIALOGUE=false)"
try:
inputs = dialogue_mod.resolve(
scene,
voices=dialogue_mod.parse_voice_map(config.DIALOGUE_VOICES),
self_voice=self._voice_id or config.ELEVENLABS_VOICE_ID,
)
except dialogue_mod.DialogueError as exc:
self.log.emit(f"dialoguectl: {exc}")
return f"[dialogue] {exc}"
text = dialogue_mod.spoken_text(scene)
self.log.emit(f"Dialogue ({len(inputs)} lines): {text[:120]}")
try:
pcm, sample_rate = tts.synthesize_dialogue(
inputs, model_id=scene.get("model"), stability=scene.get("stability")
)
except tts.TtsError as exc:
self.log.emit(f"Dialogue failed: {exc}")
# Reported, not raised: the server can read this, shorten the
# scene or fix the voice, and try again inside the same turn.
return f"[dialogue] couldn't synthesize it: {exc}"
resume = self._state.state
self._state.transition(PetState.TALKING)
self.said.emit(speech_text.for_display(text))
self.history.add(history_mod.PET, text, time.time())
should_stop = None
if self._barge_in is not None:
self._barge_in.reset()
should_stop = self._barge_in.check
completed = tts.play_pcm(pcm, sample_rate, should_stop=should_stop)
if self._barge_in is not None:
self._barge_in.reset() # the pet's own voices are in the wake window
if resume in (PetState.THINKING, PetState.IDLE):
self._state.transition(resume)
if not completed:
return dialogue_mod.describe(scene) + " (interrupted — they talked over it)"
return dialogue_mod.describe(scene)
def _apply_voice(self, reply) -> None:
"""Adopt (or drop) the voice the server tagged this reply with.
The server's `speak_as` marker names an ElevenLabs voice it just
picked — and it adds a Voice Library pick to the account first, so by
the time the id gets here it's usable for TTS. It tags *one* reply,
but the voice sticks by default: the server strips the marker before
storing the turn, so it can't recall the id later, and "keep talking
like that" would otherwise send it searching for a voice all over
again. `VOICE_STICKY=false` makes each pick last exactly one reply.
Untagged replies never *change* the voice — with stickiness on they
just keep whatever's in use, which is what makes the rest of the
conversation stay in the requested voice."""
voice_id = getattr(reply, "voice_id", "")
if voice_id:
if voice_id != self._voice_id:
self._voice_id = voice_id
self._voice_name = getattr(reply, "voice_name", "") or ""
self.log.emit(f"Voice: {self.current_voice()}")
self.voice_changed.emit(self.current_voice())
elif not config.VOICE_STICKY:
self.reset_voice()
def _speak(self, text: str, remember: bool = True) -> None:
self._state.transition(PetState.TALKING)
# Bubble gets the markdown stripped but emoji kept (it can't render
# **bold** but draws emoji fine); tts.speak() does its own, stricter
# sanitizing for the voice.
self.said.emit(speech_text.for_display(text))
self.log.emit(f"Bolt: {text}")
if remember:
self.history.add(history_mod.PET, text, time.time())
should_stop = None
@@ -250,12 +706,74 @@ class PetController(QObject):
text,
on_error=lambda exc: self.log.emit(f"TTS failed: {exc}"),
should_stop=should_stop,
voice_id=self._voice_id or None,
)
# Read the scoring history *before* resetting, or the log reports the
# blank counters instead of what actually fired.
detail = self._barge_in_detail()
if self._barge_in is not None:
# Playback fed the pet's own voice into the wake model's rolling
# window. Clear it before the idle listener starts scoring again,
# or Bolt's last sentence is still in there being re-scored.
self._barge_in.reset()
if not completed:
# You talked over it — take that as the start of the next turn
# rather than making you say the wake word again.
self.log.emit("Interrupted — listening.")
self.log.emit(f"Interrupted — listening. {detail}")
self._follow_ups = 0 # you're clearly engaged; start the count over
self._talk_now.set()
elif self._should_follow_up(text):
self._follow_ups += 1
cap = config.FOLLOW_UP_MAX_TURNS
self.log.emit(
f"Asked a question — listening for your answer "
f"({self._follow_ups}{'/' + str(cap) if cap > 0 else ''})."
)
# The tail of the reply we just played is still in the mic's ring
# buffer, and we're about to start recording with a VAD that will
# take it for the start of your answer — Bolt's own last words,
# transcribed and sent back to him as if you'd said them. Nothing you
# said can be in there: playback ran to completion, so if you had
# spoken, barge-in would have cut it and taken the other branch.
dropped = mic.flush(self._stream)
if dropped:
self.log.emit(f"Dropped {dropped} buffered frames of my own voice.")
self._pending_follow_up = True
self._talk_now.set()
def _should_follow_up(self, text: str) -> bool:
"""Whether *text* leaves the pet waiting on an answer.
Muted is excluded because mute means "don't listen to me" — an
automatic turn would walk straight past it. Napping isn't: quiet
hours suppress the pet *starting* something, and a question is only
ever asked in reply to you."""
if not config.FOLLOW_UP_LISTEN or self._muted:
return False
if not speech_text.is_question(text):
return False
# Only worth mentioning the cap on a reply that would otherwise have
# kept listening, or it fires on every statement the pet makes.
if config.FOLLOW_UP_MAX_TURNS > 0 and self._follow_ups >= config.FOLLOW_UP_MAX_TURNS:
self.log.emit("Follow-up limit reached — say the wake word to keep going.")
return False
return True
def _barge_in_detail(self) -> str:
"""Why the interruption fired, for the log. How far into playback it
happened is the tell: frame 1 means the detector was still holding
audio from before this reply started, whereas a hit several seconds
in is something the mic actually heard."""
detector = self._barge_in
if isinstance(detector, barge_in.WakeWordBargeIn):
return (
f"(wake score {detector.last_score:.3f} >= {detector.last_threshold:.2f}, "
f"peak {detector.peak_score:.3f}, at frame {detector.frames_checked} / "
f"{detector.seconds_checked:.1f}s into playback)"
)
if isinstance(detector, barge_in.BargeInDetector):
return f"(loud frames {detector.loud_frames}, threshold {config.BARGE_IN_RMS_THRESHOLD})"
return ""
# ── quiet hours / do-not-disturb ─────────────────────────────────────
@@ -298,16 +816,39 @@ class PetController(QObject):
def _queue_notification(self, notification: notifications.Notification) -> None:
"""Called on the watcher thread — just queue it; forwarding happens on
the pipeline thread where it can't collide with a live conversation."""
if not self._notification_gate.should_forward(notification, time.monotonic()):
now = time.monotonic()
if not self._notification_gate.should_forward(notification, now):
return
with self._notification_lock:
self._pending_notifications.append(notification)
if len(self._pending_notifications) == self._pending_notifications.maxlen:
# Say so rather than dropping in silence: a full queue means the
# bridge is matching more than the pet can plausibly speak, and
# the filter is what wants tightening.
self.log.emit("Notification queue full — dropping the oldest.")
self._pending_notifications.append((now, notification))
def _drain_notifications(self) -> None:
with self._notification_lock:
pending, self._pending_notifications = self._pending_notifications, []
for notification in pending:
pending = list(self._pending_notifications)
self._pending_notifications.clear()
now = time.monotonic()
max_age = config.NOTIFICATION_MAX_AGE_SECONDS
if max_age > 0:
fresh = [entry for entry in pending if now - entry[0] <= max_age]
if len(fresh) != len(pending):
self.log.emit(
f"Skipping {len(pending) - len(fresh)} notification(s) older than "
f"{int(max_age)}s."
)
pending = fresh
for index, (_stamped, notification) in enumerate(pending):
if not self._running or self._napping:
# Put back what we haven't forwarded — the old code swapped the
# queue out and then returned, silently dropping the remainder
# the moment a nap started mid-drain.
self._requeue_notifications(pending[index:])
return
self.log.emit(f"Notification: {notification.as_text()}")
self.history.add(history_mod.SYSTEM, notification.as_text(), time.time())
@@ -317,16 +858,102 @@ class PetController(QObject):
on_command=self._handle_command,
)
except server_client.ServerError as exc:
# Keep this one and everything behind it for the next heartbeat:
# the server being briefly down shouldn't silently eat the
# backlog. The age limit is what stops that retrying forever.
self.log.emit(f"Couldn't forward notification: {exc}")
self._requeue_notifications(pending[index:])
return
if reply.strip():
self._speak(reply)
self._check_deliveries()
self._apply_voice(reply)
if reply.text.strip():
self._speak(reply.text)
self._state.transition(PetState.IDLE)
def _requeue_notifications(self, entries: list) -> None:
"""Push undelivered notifications back on the front, oldest first, so a
retry keeps their original order (and their original timestamps, so a
retry loop can't keep a stale one alive indefinitely)."""
if not entries:
return
with self._notification_lock:
self._pending_notifications.extendleft(reversed(entries))
# ── file delivery ────────────────────────────────────────────────────
def _check_deliveries(self) -> None:
"""Download anything the server has queued via deliver_files —
called right after a conversation/notification turn (the common
case: "send me that file") and once per heartbeat for anything
queued out-of-band. Best-effort: a failure here is logged, not
raised, so it can't sour a turn that already got its spoken reply."""
if not config.RECEIVE_FILES:
return
try:
queued = server_client.list_outbox_files()
except server_client.ServerError as exc:
self.log.emit(f"Couldn't check for delivered files: {exc}")
return
for entry in queued:
file_id = entry.get("id")
name = entry.get("name") or file_id
if not file_id:
continue
try:
data = server_client.download_outbox_file(file_id)
except server_client.ServerError as exc:
self.log.emit(f"Couldn't download {name}: {exc}")
continue
path = file_delivery.save(config.DELIVERED_FILES_DIR, name, data)
self.log.emit(f"Received file: {path}")
# ── auto-update ──────────────────────────────────────────────────────
def _maybe_update(self) -> None:
"""Poll the Gitea releases page and, if there's a newer tag, apply it
and ask the UI to restart.
Only ever runs from the wake-listener's tick, so the pet is IDLE and
between turns by construction — an update can't land mid-sentence.
Failures are logged and the interval resets, so a server that's down
(or a release that rolls back) costs one log line an hour, not a
retry storm."""
if not config.AUTO_UPDATE or self._update_pending:
return
now = time.monotonic()
if now - self._last_update_check < config.UPDATE_CHECK_INTERVAL_SECONDS:
return
self._last_update_check = now
try:
release = updater.check_for_update()
except updater.UpdateError as exc:
self.log.emit(f"Update check failed: {exc}")
return
if release is None:
return
self.log.emit(f"Update available: {release.tag} — applying.")
try:
previous = updater.apply_update(release.tag, on_log=self.log.emit)
except updater.UpdateError as exc:
self.log.emit(f"Update to {release.tag} failed: {exc}")
return
self._update_pending = True
self.log.emit(f"Updated {previous} -> {release.tag}; restarting.")
if not self._napping:
# Napping means no proactive noise, so a silent restart it is.
self._speak(f"Updating to {release.tag}. Back in a second.")
self._state.transition(PetState.IDLE)
self.restart_requested.emit(release.tag)
# ── heartbeat ────────────────────────────────────────────────────────
def _maybe_heartbeat(self) -> None:
self._refresh_nap_state()
self._maybe_update()
if self._update_pending:
return # on the way out — don't start a conversation now
now = time.monotonic()
if now - self._last_heartbeat < config.HEARTBEAT_INTERVAL_SECONDS:
return
@@ -335,6 +962,7 @@ class PetController(QObject):
return
if self._napping:
return # quiet hours: still answers when spoken to, just doesn't start
self._check_deliveries()
self._drain_notifications()
if self._state.state != PetState.IDLE:
return
+227
View File
@@ -0,0 +1,227 @@
"""`dialoguectl` — multi-voice dialogue playback (ElevenLabs Text to Dialogue).
Normal replies are one voice saying one thing (audio/tts.py). This is the
other mode: a short *scene* two or more voices, with delivery tags the v3
model acts on (`[cheerfully]`, `[stuttering]`, `[whispering]`) synthesized
as a single take so the timing and reactions between lines actually sound
like a conversation rather than clips glued together.
Wire format, the same discipline as file_ops.py and for the same reason: it
rides the server's ordinary `command` tool marker, whose extractor only
captures up to the next newline, so the payload is a **single-line compact
JSON object**.
dialoguectl {"lines": [{"voice": "self", "text": "[cheerfully] Morning!"},
{"voice": "narrator", "text": "[whispering] He lies."}]}
The ElevenLabs field names are accepted too (`inputs` / `voice_id`), because
the model has read that API and copying its shape is the obvious thing to
try:
dialoguectl {"inputs": [{"voice_id": "9BWtsMINqrJLrRacOk9x", "text": "hi"}]}
Voices are *named*, not pasted as ids. `DIALOGUE_VOICES` in .env maps names
to ids (`narrator:9BWts,villain:IKne3`), and `self` always means the voice
the pet is speaking with right now including a voice the server picked
mid-conversation with `speak_as`, so a scene featuring Bolt sounds like
whoever Bolt currently is.
Pure parsing and validation here; the HTTP call is
`audio/tts.synthesize_dialogue` and the playback/state handling is
`controller._play_dialogue`, matching the parse/execute split used by
pet_actions.py and file_ops.py.
The API's own limits are enforced *here*, before the request goes out, so a
mistake comes back through the tool-result relay as a sentence Bolt can act
on ("too many characters, split it") rather than as an HTTP 422 he can't see.
"""
from __future__ import annotations
import json
import re
from typing import Iterable, Optional
from . import relay_json
_PREFIXES = ("dialoguectl", "dialogue", "scene")
# ElevenLabs Text to Dialogue limits (docs, 2026-07): at most 10 distinct
# voice ids per request and ~2000 characters across all inputs.
MAX_VOICES = 10
MAX_CHARS = 2000
# Names that always mean "the voice the pet is using right now".
SELF_NAMES = ("self", "bolt", "me", "pet")
# A raw ElevenLabs voice id: 20 URL-safe characters, no separators. Used to
# tell "the model pasted an id" from "the model used a name".
_VOICE_ID_RE = re.compile(r"^[A-Za-z0-9]{20}$")
class DialogueError(Exception):
"""Bad dialoguectl syntax or an unusable request — reported back to the
server as this command's output."""
def is_dialogue_command(command: str) -> bool:
parts = (command or "").strip().split(None, 1)
return bool(parts) and parts[0].lower() in _PREFIXES
def parse(command: str) -> Optional[dict]:
"""Parse `dialoguectl <json>` into {"lines": [{"voice", "text"}], ...}.
Returns None if this isn't a dialogue command at all (the caller then
tries filectl, then a real shell command). Raises DialogueError on a
dialogue command that doesn't make sense."""
if not is_dialogue_command(command):
return None
_, _, payload = (command or "").strip().partition(" ")
payload = payload.strip()
if not payload:
raise DialogueError(
'dialoguectl needs a JSON argument, e.g. dialoguectl {"lines": '
'[{"voice": "self", "text": "[cheerfully] hello"}]}'
)
# Same lenient parse as filectl (see relay_json): machine-written JSON
# fails in a handful of repeatable ways, and a stray quote shouldn't cost
# a turn — but the repair is reported back rather than hidden.
try:
data, repairs = relay_json.loads(payload)
except relay_json.RelayJsonError as exc:
raise DialogueError(
f"couldn't parse the JSON — {exc}\nIt must be one line of compact "
"JSON — put line breaks inside text as \\n, never as real newlines."
) from exc
if not isinstance(data, dict):
raise DialogueError("the argument must be a JSON object, not a list or a bare value")
raw_lines = data.get("lines")
if raw_lines is None:
raw_lines = data.get("inputs") # the ElevenLabs field name
if not isinstance(raw_lines, list) or not raw_lines:
raise DialogueError('needs a non-empty "lines" array of {"voice", "text"} objects')
lines: list[dict] = []
for index, entry in enumerate(raw_lines, start=1):
if not isinstance(entry, dict):
raise DialogueError(f"line {index} must be an object with 'voice' and 'text'")
text = str(entry.get("text") or "").strip()
if not text:
raise DialogueError(f"line {index} has no text")
voice = str(entry.get("voice") or entry.get("voice_id") or "self").strip()
lines.append({"voice": voice, "text": text})
action = {"action": "dialogue", "lines": lines}
if repairs:
action["_repairs"] = repairs
model = str(data.get("model") or data.get("model_id") or "").strip()
if model:
action["model"] = model
stability = data.get("stability")
if stability is not None:
try:
action["stability"] = min(1.0, max(0.0, float(stability)))
except (TypeError, ValueError):
raise DialogueError("stability must be a number between 0 and 1") from None
return action
def parse_voice_map(spec: str) -> dict[str, str]:
"""Parse DIALOGUE_VOICES ("narrator:9BWts…, villain:IKne3…") into a map.
Malformed entries are skipped rather than raising: a typo in .env should
cost that one voice, not the whole feature."""
voices: dict[str, str] = {}
for chunk in str(spec or "").split(","):
name, separator, voice_id = chunk.partition(":")
name, voice_id = name.strip().lower(), voice_id.strip()
if separator and name and voice_id:
voices[name] = voice_id
return voices
def resolve(
action: dict,
*,
voices: Optional[dict] = None,
self_voice: str = "",
) -> list[dict]:
"""Turn parsed lines into the API's `inputs`, resolving names to ids.
*self_voice* is the pet's current voice (which may be a `speak_as` pick,
not the configured default), so "self" tracks whoever Bolt sounds like
right now."""
known = dict(voices or {})
resolved: list[dict] = []
for index, line in enumerate(action.get("lines") or [], start=1):
name = str(line.get("voice") or "self")
key = name.lower()
if key in SELF_NAMES:
voice_id = self_voice
if not voice_id:
raise DialogueError(
"no voice is configured for the pet itself — set "
"ELEVENLABS_VOICE_ID, or name a voice from DIALOGUE_VOICES"
)
elif key in known:
voice_id = known[key]
elif _VOICE_ID_RE.match(name):
voice_id = name # a raw id pasted straight from the voice library
else:
available = ", ".join(sorted(known) + list(SELF_NAMES[:1])) or "self"
raise DialogueError(
f"line {index}: unknown voice {name!r}. Known names: {available}. "
"Use one of those, 'self' for your own voice, or a raw voice id."
)
resolved.append({"text": str(line.get("text") or ""), "voice_id": voice_id})
check_limits(resolved)
return resolved
def check_limits(inputs: Iterable[dict], *, max_voices: int = MAX_VOICES,
max_chars: int = MAX_CHARS) -> None:
"""Enforce the API's own limits before spending a request on a 422."""
entries = list(inputs)
if not entries:
raise DialogueError("no lines to speak")
distinct = {entry["voice_id"] for entry in entries}
if len(distinct) > max_voices:
raise DialogueError(
f"{len(distinct)} different voices — the limit is {max_voices} per scene"
)
total = sum(len(entry["text"]) for entry in entries)
if total > max_chars:
raise DialogueError(
f"{total} characters — the limit is {max_chars} per scene. "
"Split it into two dialoguectl calls."
)
def spoken_text(action: dict) -> str:
"""The scene as readable text, for the speech bubble and the transcript.
Delivery tags are stripped: `[cheerfully]` is a stage direction for the
model, not something to show (or, via tts.speak's sanitizer, to read out)."""
parts = []
for line in action.get("lines") or []:
text = re.sub(r"\[[^\]]{1,40}\]", " ", str(line.get("text") or ""))
text = " ".join(text.split())
if text:
parts.append(text)
return " ".join(parts)
def describe(action: dict, *, played: bool = True) -> str:
"""The tool-result string handed back to the server."""
lines = action.get("lines") or []
voices = sorted({str(line.get("voice") or "self") for line in lines})
note = relay_json.repair_note(action.get("_repairs") or [])
if not played:
return f"[dialogue] not played ({len(lines)} lines){note}"
return (
f"[dialogue] played {len(lines)} line{'s' if len(lines) != 1 else ''} "
f"in {len(voices)} voice{'s' if len(voices) != 1 else ''}: {', '.join(voices)}{note}"
)
+54
View File
@@ -0,0 +1,54 @@
"""Saves files the server queues via its deliver_files tool (ai/desk_api.py
in the main tmn-api repo) to a local downloads folder.
The server side of this is already generic any desk client can list
GET /desk/files and fetch GET /desk/files/<id> (see server_client.
list_outbox_files / download_outbox_file) so this module is just the
filesystem half: turn a server-supplied display name into a safe path and
write the bytes.
Pure filename/path logic lives here so it's testable without touching a real
mic/network; the only I/O is the final write in save().
"""
from __future__ import annotations
from pathlib import Path
_FALLBACK_NAME = "delivered_file"
def sanitize_filename(name: str) -> str:
"""Reduce a server-supplied name to a bare filename. Defends against a
delivered name that's actually a path (../../etc, an absolute path, ...)
Path(...).name strips every directory component, and anything that
collapses to nothing (or "." / "..") falls back to a generic name."""
candidate = Path(str(name or "").strip()).name
if candidate in ("", ".", ".."):
return _FALLBACK_NAME
return candidate
def unique_path(directory: Path, name: str) -> Path:
"""*name* under *directory*, suffixed " (1)", " (2)", ... if that name is
already taken a delivered file never overwrites an earlier download."""
directory = Path(directory)
directory.mkdir(parents=True, exist_ok=True)
candidate = directory / name
if not candidate.exists():
return candidate
stem, suffix = candidate.stem, candidate.suffix
n = 1
while True:
candidate = directory / f"{stem} ({n}){suffix}"
if not candidate.exists():
return candidate
n += 1
def save(directory: Path, name: str, data: bytes) -> Path:
"""Write *data* under *directory* as *name* (sanitized + uniquified),
returning the path written."""
path = unique_path(directory, sanitize_filename(name))
path.write_bytes(data)
return path
+290
View File
@@ -0,0 +1,290 @@
"""Local file read/edit/write, intercepted from the server-relayed command
channel the same way pet_actions.py intercepts petctl (see server_client.
run_local_command / controller._handle_command). Whatever text the server's
"command" tool sends is just a string this repo is free to interpret before
it ever reaches subprocess a `filectl` pseudo-command is one such
interpretation, giving the model a way to read/write/edit files on this
machine without constructing a raw shell heredoc, where quoting, `$`,
backticks, and embedded quotes make anything beyond a one-liner failure-prone.
Executing arbitrary commands already works today (that's exactly what
run_local_command/subprocess.run does) filectl doesn't add that ability,
it only makes the read/write/edit slice of it reliable. It also doesn't
expand what the server can already do to this machine: a relayed shell
command could already overwrite any file the desktop user can write (see the
security notes in CLAUDE.md) filectl is a safer *path* to the same
capability, not a new capability.
Wire format: `filectl <json>`, where <json> is a single-line, compact JSON
object critically, ONE LINE. The server's "command" tool marker only
captures the argument up to the next newline (see TOOL_SPECS/
_extract_all_tool_calls in the main repo's ai/agents/default.py — "command"
is not declared multiline), so a marker-delimited multi-line payload (this
module's first design) silently got truncated at the first line no matter
how the prompt worded it. JSON sidesteps that for free: json.dumps() already
encodes embedded newlines as the two characters "\n", not an actual line
break, so arbitrarily multi-line file content still fits on the one physical
line the extractor captures.
filectl {"op": "list", "path": "<dir>", "pattern": "<glob>", "recursive": <bool>}
filectl {"op": "read", "path": "<path>", "start": <line>, "end": <line>}
filectl {"op": "write", "path": "<path>", "content": "<text>"}
filectl {"op": "edit", "path": "<path>", "old": "<text>", "new": "<text>"}
"pattern"/"recursive" (list) and "start"/"end" (read) are optional. `list`
exists even though a real `ls`/`dir` shell command already works, because
this repo is cross-platform (Windows/macOS/Linux) and the model shouldn't
have to guess which listing command applies on this machine one glob-based
op covers all three. Must be invoked as the argument to the
ordinary `command` tool marker (e.g. `command: filectl {"op": "write", ...}`)
see the pet-only paragraph in the main repo's ai/desk_api.py
_system_context for the exact instruction the model is given, including the
"keep it one line" requirement.
Pure parsing (parse) is separated from the filesystem I/O (execute) so the
syntax is unit-testable without touching disk, matching pet_actions.py's
parse/describe split.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Optional
from . import relay_json
_PREFIXES = ("filectl", "file")
# Keeps a runaway read/write/list from blowing up the tool_result relay (and,
# for read/list, from flooding the model's context with a giant response).
_MAX_READ_BYTES = 200_000
_MAX_WRITE_BYTES = 2_000_000
_MAX_LIST_ENTRIES = 500
HELP = (
'filectl {"op": "list", "path": "<dir>", "pattern": "<glob>", "recursive": <bool>}\n'
'filectl {"op": "read", "path": "<path>", "start": <line>, "end": <line>}\n'
'filectl {"op": "write", "path": "<path>", "content": "<text>"}\n'
'filectl {"op": "edit", "path": "<path>", "old": "<text>", "new": "<text>"}\n'
"Must be all on one line — this rides the single-line \"command\" marker."
)
class FileOpError(Exception):
"""Bad filectl syntax or a filesystem error — reported back to the
server as command output, exactly like ActionError in pet_actions.py."""
def is_file_command(command: str) -> bool:
stripped = (command or "").strip()
if not stripped:
return False
first_word = stripped.split(None, 1)[0]
return first_word.lower() in _PREFIXES
def parse(command: str) -> Optional[dict]:
"""Parse a `filectl <json>` string into an action dict, or None if this
isn't a filectl command at all (caller should try the next handler, or
fall back to a real shell command). Raises FileOpError on a filectl
command that doesn't make sense."""
if not is_file_command(command):
return None
stripped = command.strip()
_, _, rest = stripped.partition(" ")
rest = rest.strip()
if not rest or rest.lower() in ("help", "-h", "--help"):
return {"action": "help"}
# Lenient on purpose — see relay_json. A stray quote in machine-written
# JSON should not cost a turn, but the repair is reported back so the model
# is told it sent something broken while it can still learn from it.
try:
payload, repairs = relay_json.loads(rest)
except relay_json.RelayJsonError as exc:
raise FileOpError(f"couldn't parse filectl JSON — {exc}\nusage:\n{HELP}") from exc
if not isinstance(payload, dict):
raise FileOpError(f"filectl payload must be a JSON object; usage:\n{HELP}")
op = str(payload.get("op") or "help").lower()
# Carried on the action so execute() can tell the model what it got wrong;
# a silent repair would fix today's call and guarantee tomorrow's.
tag = {"_repairs": repairs} if repairs else {}
if op == "help":
return {"action": "help", **tag}
if op == "list":
path = _required_str(payload, "path")
pattern = payload.get("pattern", "*")
if not isinstance(pattern, str) or not pattern:
raise FileOpError('"pattern" must be a non-empty string')
return {
"action": "list", "path": path,
"pattern": pattern, "recursive": bool(payload.get("recursive")), **tag,
}
if op == "read":
path = _required_str(payload, "path")
return {
"action": "read", "path": path,
"start": _line_number(payload.get("start"), "start"),
"end": _line_number(payload.get("end"), "end"), **tag,
}
if op == "write":
path = _required_str(payload, "path")
if payload.get("content") is None:
raise FileOpError('write needs "content"')
return {"action": "write", "path": path, "content": str(payload["content"]), **tag}
if op == "edit":
path = _required_str(payload, "path")
if payload.get("old") is None or payload.get("new") is None:
raise FileOpError('edit needs "old" and "new"')
old, new = str(payload["old"]), str(payload["new"])
if old == new:
raise FileOpError("old and new text are identical — nothing to edit")
return {"action": "edit", "path": path, "old": old, "new": new, **tag}
raise FileOpError(f"unknown filectl op {op!r}; usage:\n{HELP}")
def _required_str(payload: dict, key: str) -> str:
value = payload.get(key)
if not isinstance(value, str) or not value.strip():
raise FileOpError(f'filectl needs a non-empty "{key}"')
return value
def _line_number(value, label: str) -> Optional[int]:
if value is None:
return None
if isinstance(value, bool) or not isinstance(value, int):
raise FileOpError(f"{label} must be an integer line number, got {value!r}")
return value
def describe(action: dict) -> str:
"""Text handed back to the server before execute() runs — mirrors
pet_actions.describe, used only for the log line."""
kind = action.get("action")
if kind == "help":
return "[filectl] help"
return f"[filectl] {kind} {action.get('path', '')}"
def execute(action: dict) -> str:
"""Actually perform a parsed filectl action, returning the text to send
back to the server as the command's output. Raises FileOpError on any
filesystem problem, same as a bad-syntax parse error."""
kind = action.get("action")
if kind == "help":
return HELP
if kind == "list":
output = _do_list(action)
elif kind == "read":
output = _do_read(action)
elif kind == "write":
output = _do_write(action)
elif kind == "edit":
output = _do_edit(action)
else:
output = "[filectl] ok"
return output + relay_json.repair_note(action.get("_repairs") or [])
def _resolve(path_str: str) -> Path:
return Path(path_str).expanduser()
def _do_list(action: dict) -> str:
path = _resolve(action["path"])
if not path.is_dir():
raise FileOpError(f"no such directory: {path}")
pattern = action["pattern"]
glob = path.rglob if action["recursive"] else path.glob
try:
entries = sorted(glob(pattern), key=lambda p: str(p).lower())
except OSError as exc:
raise FileOpError(f"couldn't list {path}: {exc}") from exc
if not entries:
return f"[no entries matching {pattern!r} in {path}]"
truncated = len(entries) > _MAX_LIST_ENTRIES
lines = []
for entry in entries[:_MAX_LIST_ENTRIES]:
rel = entry.relative_to(path)
if entry.is_dir():
lines.append(f"{rel}/")
continue
try:
size = entry.stat().st_size
except OSError:
size = -1
lines.append(f"{rel}\t{size}B")
if truncated:
lines.append(f"... truncated at {_MAX_LIST_ENTRIES} entries (of {len(entries)}) — narrow \"pattern\"")
return "\n".join(lines)
def _do_read(action: dict) -> str:
path = _resolve(action["path"])
if not path.is_file():
raise FileOpError(f"no such file: {path}")
try:
data = path.read_bytes()
except OSError as exc:
raise FileOpError(f"couldn't read {path}: {exc}") from exc
if len(data) > _MAX_READ_BYTES:
raise FileOpError(
f"{path} is {len(data)} bytes, over the {_MAX_READ_BYTES}-byte filectl read limit — "
'pass "start"/"end" to read a slice instead'
)
try:
text = data.decode("utf-8")
except UnicodeDecodeError as exc:
raise FileOpError(f"{path} isn't valid UTF-8 text: {exc}") from exc
lines = text.splitlines()
start = max(1, action.get("start") or 1)
end = min(len(lines), action.get("end") or len(lines))
if not lines:
return "[empty file]"
return "\n".join(f"{i:>6}\t{lines[i - 1]}" for i in range(start, end + 1))
def _do_write(action: dict) -> str:
path = _resolve(action["path"])
content = action["content"]
if len(content.encode("utf-8")) > _MAX_WRITE_BYTES:
raise FileOpError(f"content is over the {_MAX_WRITE_BYTES}-byte filectl write limit")
try:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8")
except OSError as exc:
raise FileOpError(f"couldn't write {path}: {exc}") from exc
return f"[filectl] wrote {len(content)} chars to {path}"
def _do_edit(action: dict) -> str:
path = _resolve(action["path"])
old, new = action["old"], action["new"]
if not path.is_file():
raise FileOpError(f"no such file: {path}")
try:
text = path.read_text(encoding="utf-8")
except OSError as exc:
raise FileOpError(f"couldn't read {path}: {exc}") from exc
count = text.count(old)
if count == 0:
raise FileOpError(f"didn't find that text in {path} — nothing changed")
if count > 1:
raise FileOpError(
f"that text appears {count} times in {path} — filectl edit needs a unique match; "
"include more surrounding context"
)
try:
path.write_text(text.replace(old, new, 1), encoding="utf-8")
except OSError as exc:
raise FileOpError(f"couldn't write {path}: {exc}") from exc
return f"[filectl] edited {path}"
+206
View File
@@ -0,0 +1,206 @@
"""Things you say to the pet that the server has no business answering.
"stop", "come here", "go to sleep", "say that again", "use your normal voice"
none of these are questions for Bolt's brain. They're commands to the *body*,
and today every one of them costs a full turn: Deepgram, a `/desk/converse`
round trip, a model deciding to emit `petctl`, then ElevenLabs. Two to four
seconds and three network hops to make the pet walk left, and it only works at
all if the server's prompt happens to advertise the right verb — which is
exactly why `petctl voice reset` needs a block in the server's pet prompt (see
CLAUDE.md) or the model never emits it. Recognising the phrase here removes
both the latency and that coupling: "go back to your normal voice" works
whether or not the server was ever told the voice can be reset.
The whole design problem is **not stealing real requests**. Three rules keep
it honest:
1. **Whole-utterance, exact match after normalisation.** Never substring. So
"stop" is an intent and "stop the docker container" is a question for the
server the distinction a substring match would destroy.
2. **The phrase table is closed and small.** Every entry is something with no
plausible reading as a request for Bolt to *do work*. Anything arguable
("no thanks", "nothing") is deliberately absent see rule 3 for why a
wrong guess is expensive.
3. **Nothing is recognised mid-conversation.** The controller skips this
entirely on a follow-up turn: if Bolt just asked you something, your answer
belongs to him, and swallowing "never mind" locally would leave the server
holding a question it never got an answer to. Local intents are only ever
for turns *you* started.
Both sides of the comparison go through `normalize()` the table is
canonicalised at import so phrases can be written the way a person says them
("go back to your normal voice") without every variant having to be spelled
out. Filler is dropped from anywhere, not just the ends, because STT scatters
it ("hey bolt, could you please just stop now").
Pure classification, like pet_actions.parse: this module decides *what was
meant* and hands back an action in the same shape pet_actions produces, so
`controller.action` and `PetWindow.apply_action` need no new vocabulary. The
effects live in controller._handle_local_intent.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Optional
# Words with no bearing on any command in the table, dropped wherever they
# appear. Kept deliberately short: every entry here is a word that can't
# distinguish one of these phrases from another, and adding one that can is how
# two intents quietly collide (the builder below raises if that happens).
_FILLER = frozenset({
"the", "a", "an", "my", "your", "yours", "its", "to", "of", "and",
"please", "just", "that", "some", "bolt", "thunderbolt", "pet", "buddy",
})
# Dropped only from the front — the politeness/address ramp STT reliably
# prefixes. Not safe to drop mid-phrase (a bare "do" or "go" carries meaning
# elsewhere), which is why this is separate from _FILLER.
_LEADING_FILLER = frozenset({
"hey", "hi", "hello", "yo", "ok", "okay", "um", "uh", "er", "so",
"can", "could", "would", "will", "you", "i", "id", "like", "lets",
"let", "us", "do", "go", "then", "now",
})
_TRAILING_FILLER = frozenset({
"ok", "okay", "thanks", "thank", "you", "boy", "already", "now",
})
_KEEP = re.compile(r"[^a-z0-9 ]+")
def normalize(text: str) -> str:
"""Reduce an utterance to the bare command, or "" if nothing is left.
Lowercase, punctuation stripped (STT punctuates inconsistently), filler
dropped. Not a stemmer and deliberately not clever its only job is to
make the same command spoken two ways land on the same string, without
ever turning one command into a different one."""
words = [word for word in _KEEP.sub(" ", (text or "").lower()).split()
if word not in _FILLER]
while words and words[0] in _LEADING_FILLER:
words.pop(0)
while words and words[-1] in _TRAILING_FILLER:
words.pop()
return " ".join(words)
@dataclass(frozen=True)
class Intent:
"""One recognised local command.
*action* is a pet_actions-shaped dict for the UI (or None when there's
nothing for the body to do); *speak* is what to say out loud, empty for the
intents where doing the thing silently *is* the acknowledgement the pet
visibly moves, and a spoken confirmation would only make it slower. "stop"
in particular has to be silent: answering "okay!" when told to be quiet is
a comedy sketch, not a feature.
"""
name: str
action: Optional[dict] = None
speak: str = ""
# Intent -> (the Intent, the phrases that mean it, written as spoken).
_TABLE: tuple[tuple[Intent, tuple[str, ...]], ...] = (
(
Intent("stop"),
("stop", "stop talking", "stop it", "be quiet", "quiet", "shut up",
"hush", "never mind", "nevermind", "forget it", "cancel",
"cancel that", "drop it", "enough"),
),
(
Intent("nap", {"action": "nap", "enabled": True}, "Night."),
("go to sleep", "take a nap", "have a nap", "go to bed", "bedtime",
"goodnight", "good night", "get some rest"),
),
(
Intent("wake", {"action": "nap", "enabled": False}, "I'm up."),
("wake up", "get up", "rise and shine", "you're awake", "are you awake"),
),
(
Intent("come", {"action": "move", "anchor": "cursor"}),
("come here", "come to me", "come back", "over here", "follow me",
"follow my cursor"),
),
(
Intent("go_away", {"action": "move", "anchor": "bottom-right"}),
("go away", "move over", "move out of the way", "get out of the way",
"out of the way", "hide", "get lost", "shoo", "scram",
"go somewhere else"),
),
(
Intent("repeat"), # answered from history by the controller
("say that again", "say again", "repeat that", "repeat",
"what did you say", "what was that", "come again", "one more time",
"again", "sorry what"),
),
(
Intent("wander_on", {"action": "wander", "enabled": True}),
("go for a walk", "wander", "wander around", "walk around", "explore",
"stretch your legs", "roam"),
),
(
Intent("wander_off", {"action": "wander", "enabled": False}),
("stay still", "stay put", "stop moving", "stop wandering",
"don't move", "sit", "sit still", "stay", "settle down", "hold still"),
),
(
# Reachable from the server too (petctl voice reset), but only if its
# prompt mentions the verb. Recognising it here is what makes the
# phrase work regardless of what the server was told.
Intent("voice_reset", None, "Back to my own voice."),
("use your normal voice", "use your own voice", "your normal voice",
"go back to your normal voice", "be yourself", "be yourself again",
"stop doing that voice", "drop the voice", "talk normally",
"speak normally", "use your real voice"),
),
)
def _build() -> dict[str, Intent]:
"""Canonicalise the table, refusing to build an ambiguous one.
A phrase that normalises to "" would match an utterance of pure filler
("hey bolt"), and one that lands on the same string as a phrase from
another intent would silently bind to whichever was declared last. Both are
edit-time mistakes, so they fail at import rather than at 3am on a mic."""
table: dict[str, Intent] = {}
for intent, phrases in _TABLE:
for phrase in phrases:
key = normalize(phrase)
if not key:
raise ValueError(f"intent phrase {phrase!r} normalises to nothing")
existing = table.get(key)
if existing is not None and existing.name != intent.name:
raise ValueError(
f"phrase {phrase!r} ({key!r}) is claimed by both "
f"{existing.name} and {intent.name}"
)
table[key] = intent
return table
_BY_PHRASE = _build()
# Longest phrase in the table, in words. Anything longer can't match, so a real
# request skips normalisation entirely — this runs on every turn.
_MAX_WORDS = max(len(phrase.split()) for phrase in _BY_PHRASE)
def recognize(text: str) -> Optional[Intent]:
"""The intent *text* expresses, or None to send it to the server.
None is the safe answer and the common one: anything not matched verbatim
against the table belongs to Bolt."""
raw = (text or "").strip()
if not raw:
return None
# +6 words of slack for the filler about to be stripped ("hey bolt, could
# you please stop" is six words to reach a one-word command).
if len(raw.split()) > _MAX_WORDS + 6:
return None
intent = _BY_PHRASE.get(normalize(raw))
return intent
+213
View File
@@ -0,0 +1,213 @@
"""Which screens exist, and which one the pet is standing on.
Deliberately free of Qt *and* of any subprocess probing: the monitor list is
published by the UI (`ui/pet_window.py` builds it from
`QGuiApplication.screens()`) and handed to the controller over a queued
signal, the same way every other UIcontroller message travels.
That indirection is the whole point. The pet has to agree with itself about
what "monitor 2" means if the controller enumerated screens with `xrandr`
while the window jumped using Qt's screen list, the two orderings could
disagree and Bolt would announce one screen and land on another. Making Qt the
single source of truth removes that class of bug, and leaves everything here
pure enough to unit test without a display.
Indices are **1-based in every string a human or the model ever sees**, and
0-based in the list itself. `Monitor.index` is the 0-based one; `.number` is
what gets printed.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Iterable, Optional
# Directional specs understood by resolve(), mapped to a (dx, dy) heading.
_DIRECTIONS = {
"left": (-1, 0),
"right": (1, 0),
"up": (0, -1),
"above": (0, -1),
"down": (0, 1),
"below": (0, 1),
}
@dataclass(frozen=True)
class Monitor:
"""One screen, in the global desktop coordinate space."""
index: int # 0-based position in the published list
name: str
x: int
y: int
width: int
height: int
primary: bool = False
@property
def number(self) -> int:
"""1-based, for anything a person or the model reads."""
return self.index + 1
@property
def right(self) -> int:
return self.x + self.width
@property
def bottom(self) -> int:
return self.y + self.height
@property
def center(self) -> tuple[int, int]:
return self.x + self.width // 2, self.y + self.height // 2
def contains(self, x: int, y: int) -> bool:
return self.x <= x < self.right and self.y <= y < self.bottom
@property
def label(self) -> str:
bits = f"{self.number}: {self.name} {self.width}x{self.height}"
return bits + " (primary)" if self.primary else bits
def monitor_containing(
monitors: Iterable[Monitor], x: int, y: int
) -> Optional[Monitor]:
"""The screen holding point (x, y), or None if it's off every screen."""
for monitor in monitors:
if monitor.contains(x, y):
return monitor
return None
def nearest_monitor(monitors: Iterable[Monitor], x: int, y: int) -> Optional[Monitor]:
"""Screen whose centre is closest to (x, y) — the fallback when a point
lands in the dead space between mismatched screens."""
monitors = list(monitors)
if not monitors:
return None
return min(
monitors,
key=lambda m: (m.center[0] - x) ** 2 + (m.center[1] - y) ** 2,
)
def resolve(
monitors: list[Monitor], spec: str, current: Optional[int] = None
) -> Monitor:
"""Turn a `petctl jump` target into a screen.
Accepts a 1-based number, a name (case-insensitive substring, so "hdmi"
finds "HDMI-0"), `next`/`prev`, `primary`, `other`, or a direction
(`left`/`right`/`up`/`down`) relative to *current*. Raises ValueError with
a message meant to be read by the model, since it goes back as tool
output.
"""
if not monitors:
raise ValueError("no monitors have been reported yet")
spec = (spec or "").strip().lower()
if not spec:
raise ValueError("jump needs a target monitor")
count = len(monitors)
if current is None or not (0 <= current < count):
current = next((m.index for m in monitors if m.primary), 0)
if spec.isdigit():
number = int(spec)
if not (1 <= number <= count):
raise ValueError(
f"there is no monitor {number}; you have {count} "
f"(1-{count})"
)
return monitors[number - 1]
if spec in ("next", "forward"):
return monitors[(current + 1) % count]
if spec in ("prev", "previous", "back"):
return monitors[(current - 1) % count]
if spec == "primary":
return next((m for m in monitors if m.primary), monitors[0])
if spec == "other":
# With two screens "the other one" is unambiguous; with more it's just
# the next one round, which is at least always a *different* screen.
return monitors[(current + 1) % count]
if spec == "random":
# Deterministic-free choice is the caller's business; pick the screen
# furthest from the current one so "random" always visibly moves.
here = monitors[current].center
return max(
monitors,
key=lambda m: (m.center[0] - here[0]) ** 2 + (m.center[1] - here[1]) ** 2,
)
if spec in _DIRECTIONS:
dx, dy = _DIRECTIONS[spec]
here = monitors[current].center
candidates = []
for monitor in monitors:
if monitor.index == current:
continue
ox, oy = monitor.center
along = (ox - here[0]) * dx + (oy - here[1]) * dy
if along <= 0:
continue # not in that direction at all
drift = abs((ox - here[0]) * dy + (oy - here[1]) * dx)
candidates.append((drift, along, monitor))
if not candidates:
raise ValueError(
f"there's no monitor to the {spec} of monitor "
f"{monitors[current].number}"
)
# Prefer the best-aligned screen, then the closest of those.
candidates.sort(key=lambda item: (item[0], item[1]))
return candidates[0][2]
matches = [m for m in monitors if spec in m.name.lower()]
if len(matches) == 1:
return matches[0]
if len(matches) > 1:
raise ValueError(
f"{spec!r} matches several monitors: "
+ ", ".join(m.label for m in matches)
)
raise ValueError(
f"unknown monitor {spec!r}; you have: "
+ "; ".join(m.label for m in monitors)
+ " — or use next/prev/primary/left/right/up/down"
)
def describe(monitors: list[Monitor], current: Optional[int] = None) -> str:
"""Full listing, used as `petctl monitors` output."""
if not monitors:
return "[pet] no monitor information available"
lines = [f"[pet] {len(monitors)} monitor(s):"]
for monitor in monitors:
here = " <- Bolt is here" if monitor.index == current else ""
lines.append(
f" {monitor.label} at +{monitor.x}+{monitor.y}{here}"
)
return "\n".join(lines)
def summary(monitors: list[Monitor], current: Optional[int] = None) -> Optional[str]:
"""One-line version tacked onto each utterance — short on purpose, since
it rides along with every single thing you say."""
if not monitors:
return None
parts = ", ".join(f"{m.number}) {m.name} {m.width}x{m.height}" for m in monitors)
line = f"{len(monitors)} monitors: {parts}"
if current is not None and 0 <= current < len(monitors):
line += f"; Bolt is on {monitors[current].number}"
return line
def annotate(text: str, monitors: list[Monitor], current: Optional[int] = None) -> str:
"""Attach the screen summary as a separate aside, matching the style of
screen_context.annotate() so the model can ignore it when irrelevant."""
text = (text or "").strip()
line = summary(monitors, current)
if not text or not line:
return text
return f"{text}\n\n[{line}]"
+55 -1
View File
@@ -29,14 +29,32 @@ ANCHORS = (
EMOTES = ("wave", "hop", "spin", "nod", "shake", "bounce", "wiggle")
# Where `petctl jump` can be aimed. A bare number (1-based) works too, as does
# any unique part of a monitor's name — resolution lives in monitors.resolve().
MONITOR_SPECS = (
"next", "prev", "primary", "other", "random",
"left", "right", "up", "down",
)
HELP = (
"petctl move <x> <y> | <" + "|".join(ANCHORS) + ">\n"
"petctl jump <monitor number|" + "|".join(MONITOR_SPECS) + "|name>\n"
"petctl monitors\n"
"petctl read [monitor number|here|all]\n"
"petctl emote <" + "|".join(EMOTES) + ">\n"
"petctl say <text>\n"
"petctl wander on|off\n"
"petctl nap on|off"
"petctl nap on|off\n"
"petctl voice reset\n"
"petctl self_restart [why]"
)
# `petctl voice` only ever goes one way: back to the configured voice. Picking
# a *different* one is the server's job (its speak_as reply marker), and it
# already knows how — what it has no way to say is "never mind, be yourself
# again", because it was never told which voice that is.
VOICE_RESETS = ("reset", "default", "normal", "own", "back", "mine", "yours")
class ActionError(Exception):
"""Bad petctl syntax — reported back to the server as command output."""
@@ -82,6 +100,24 @@ def parse(command: str) -> Optional[dict]:
raise ActionError(f"unknown position {args[0]!r}; try one of: " + ", ".join(ANCHORS))
return {"action": "move", "anchor": anchor}
if verb in ("jump", "monitor", "screen"):
if not args:
raise ActionError(
"jump needs a monitor: a number, a name, or one of "
+ ", ".join(MONITOR_SPECS)
)
# The spec isn't validated here on purpose: which monitors exist is a
# runtime fact this pure module doesn't have. monitors.resolve() does
# it once the published screen list is in hand.
return {"action": "jump", "target": " ".join(args).strip()}
if verb in ("monitors", "screens", "displays"):
return {"action": "monitors"}
if verb in ("read", "look", "ocr", "see"):
target = (" ".join(args).strip() or "here").lower()
return {"action": "read", "target": target}
if verb in ("emote", "do"):
if not args:
raise ActionError("emote needs a name: " + ", ".join(EMOTES))
@@ -101,6 +137,22 @@ def parse(command: str) -> Optional[dict]:
raise ActionError("wander needs on or off")
return {"action": "wander", "enabled": _bool_arg(args[0])}
if verb == "voice":
target = (args[0].lower() if args else "reset").lstrip("-")
if target not in VOICE_RESETS:
raise ActionError(
f"can't set a voice from petctl (got {args[0]!r}); "
"use the speak_as reply marker to pick one. "
"petctl voice reset goes back to the default voice."
)
return {"action": "voice", "voice": "default"}
if verb in ("self_restart", "restart", "reboot"):
# Free text, not a fixed grammar: the argument is a note to the pet's
# *next* process about why it died and what to look at when it comes
# back, so anything the model wants to tell future-itself is valid.
return {"action": "self_restart", "reason": " ".join(args).strip()}
if verb in ("nap", "sleep", "dnd"):
if not args:
raise ActionError("nap needs on or off")
@@ -124,6 +176,8 @@ def describe(action: dict) -> str:
if kind == "move":
where = action.get("anchor") or f"({action.get('x')}, {action.get('y')})"
return f"[pet] walking to {where}"
if kind == "jump":
return f"[pet] jumping to monitor {action['target']}"
if kind == "emote":
return f"[pet] {action['emote']}"
if kind == "say":
+148
View File
@@ -0,0 +1,148 @@
"""Lenient JSON for the relayed command channel — and honest about it.
`filectl` and `dialoguectl` both take a single line of compact JSON, hand-typed
by a language model into a tool marker. Models get that *nearly* right and then
get it wrong in a small, boringly repeatable set of ways:
{"op":"list","path":"/home/x","recursive":false"} ← stray quote after a literal
{"op": "read", "path": "/tmp/a.txt",} trailing comma
{'op': 'read', 'path': '/tmp/a.txt'} single quotes
{op: read, path: /tmp/a.txt} smart quotes
{"op": "list", "recursive": False} Python literals
```json {"op": "list"} ``` fenced
Observed live (2026-07-30): a stray quote after `false` cost an entire desk
turn the call was rejected, the model re-sent the *identical* line, was
rejected again, and then gave up and told the user "I'll check now" without
ever calling anything. The user got a promise instead of an answer because of
one character.
Strict parsing is the wrong trade here. Nothing about a misplaced quote is
ambiguous, the payload is machine-written and machine-read, and the cost of
refusing is a wasted round trip that the model has already demonstrated it
won't recover from. So: try strict first, then apply narrow repairs, and
accept a repair **only if the result parses**.
Two rules keep this from becoming "guess what they meant":
1. **Repairs are conservative and named.** Each one fixes a known malformation,
is applied in isolation, and is reported by name.
2. **Repairs are never silent.** The caller appends the repair note to the tool
result, so the model is told it sent broken JSON *while it still has the
turn* the fix works today and teaches within the conversation. Hiding it
would trade a visible failure for an invisible one.
When nothing parses, the error points at the exact character with a caret,
because "Expecting ',' delimiter: char 74" is not something a model can act on
and a pointed-at fragment is.
"""
from __future__ import annotations
import json
import re
from typing import Any, Callable
# Ordered, cheapest and safest first. Each entry is (name, transform); after
# each one the payload is re-parsed, so the first repair that works wins and
# nothing more aggressive gets applied than the input actually needed.
_REPAIRS: tuple[tuple[str, Callable[[str], str]], ...] = (
(
"stripped a markdown code fence",
lambda text: re.sub(r"^\s*```(?:json)?\s*|\s*```\s*$", "", text),
),
(
"replaced smart quotes with straight ones",
lambda text: text.translate(str.maketrans({"": '"', "": '"',
"": "'", "": "'"})),
),
(
"removed a stray quote after a bare value",
# {"recursive":false"} -> {"recursive":false}
lambda text: re.sub(
r"(:\s*(?:true|false|null|-?\d+(?:\.\d+)?))\s*\"(\s*[,}\]])", r"\1\2", text),
),
(
"removed a trailing comma",
lambda text: re.sub(r",(\s*[}\]])", r"\1", text),
),
(
"converted Python literals (True/False/None) to JSON",
lambda text: re.sub(r"(:\s*)(True|False|None)\b",
lambda m: m.group(1) + {"True": "true", "False": "false",
"None": "null"}[m.group(2)], text),
),
(
"converted single-quoted strings to double-quoted",
lambda text: re.sub(r"'([^'\"]*)'", r'"\1"', text),
),
)
class RelayJsonError(ValueError):
"""Unparseable even after repairs — carries a pointed-at fragment."""
def loads(payload: str) -> tuple[Any, list[str]]:
"""Parse *payload*, repairing common model mistakes.
Returns (data, repairs) where *repairs* names what had to be fixed empty
when the input was already valid. Raises RelayJsonError with a caret at the
offending character when nothing works."""
text = str(payload or "").strip()
if not text:
raise RelayJsonError("empty payload")
try:
return json.loads(text), []
except json.JSONDecodeError as exc:
# Bound to a plain name: Python deletes the `as` target at the end of
# the except block, so referring to it further down would raise
# UnboundLocalError instead of reporting the parse failure.
first_error = exc
applied: list[str] = []
candidate = text
for name, repair in _REPAIRS:
repaired = repair(candidate)
if repaired == candidate:
continue
candidate = repaired
applied.append(name)
try:
return json.loads(candidate), applied
except json.JSONDecodeError:
continue # keep going: a payload can be broken in more than one way
raise RelayJsonError(point_at(text, first_error))
def point_at(text: str, error: json.JSONDecodeError, width: int = 28) -> str:
"""Show the failure where it happened.
A model can act on "you wrote `false\"}` here"; it cannot act on
"Expecting ',' delimiter: line 1 column 75"."""
position = max(0, min(len(text), getattr(error, "pos", 0)))
start = max(0, position - width)
end = min(len(text), position + width)
fragment = text[start:end]
caret = " " * (position - start) + "^"
lead = "" if start > 0 else ""
tail = "" if end < len(text) else ""
return (
f"{error.msg} at character {position}:\n"
f" {lead}{fragment}{tail}\n"
f" {' ' * len(lead)}{caret}"
)
def repair_note(repairs: list[str]) -> str:
"""The line appended to a tool result when repairs were needed.
Phrased as feedback rather than an apology: the model is the author of the
broken JSON and is the one who can stop sending it."""
if not repairs:
return ""
return (
" (note: your JSON was malformed — I " + "; ".join(repairs)
+ " and ran it anyway. Send valid single-line JSON next time.)"
)
+187
View File
@@ -0,0 +1,187 @@
"""Reading the text that's actually on a monitor, via screenshot + OCR.
This is the "Bolt can see what's on screen" half of the screen features. It is
**pull, not push**: nothing here runs on its own. The server has to ask, by
relaying `petctl read`, and the recognised text goes back as that command's
output through the existing tool-result relay (see server_client.converse).
That's deliberate on two counts — OCR of a 4K screen costs a second or two,
which would be tacked onto every single utterance if it ran automatically, and
"screen contents leave this machine" should be a thing Bolt decides to do and
you can see in the log, not a silent constant.
Both halves are optional and soft-fail with a reason, the way hotkey.py does:
capture needs `mss`, recognition needs a Tesseract or RapidOCR install. With
neither, `petctl read` reports what's missing instead of raising, and the rest
of the pet carries on.
The pure parts (cleaning OCR output, formatting the reply, deciding which
engine to use given what's installed) are split out and unit tested; only
capture and the OCR call itself need a real screen.
"""
from __future__ import annotations
import re
import shutil
from typing import Callable, Optional
from .monitors import Monitor
DEFAULT_MAX_CHARS = 4000
INSTALL_HINT = (
"install one of: `pip install mss pytesseract` + `sudo apt install "
"tesseract-ocr` (fastest), or `pip install mss rapidocr-onnxruntime` "
"(no system package needed)"
)
# Lines that are almost certainly OCR noise rather than text: window chrome
# fragments, isolated punctuation, single stray characters.
_MIN_MEANINGFUL = 2
def _module_available(name: str) -> bool:
import importlib.util
try:
return importlib.util.find_spec(name) is not None
except (ImportError, ValueError):
return False
# ── pure helpers (unit tested; no screen, no OCR engine needed) ──────────────
def resolve_engine(
has_module: Callable[[str], bool] = _module_available,
which: Callable[[str], Optional[str]] = shutil.which,
) -> tuple[Optional[str], str]:
"""Pick an OCR engine from what's installed.
Returns `(engine, reason)`. *engine* is None when nothing usable is
present, and *reason* then explains what to install. Probes are injected
so this is testable on a machine with a different set of things installed.
"""
if has_module("pytesseract") and which("tesseract"):
return "pytesseract", ""
if has_module("rapidocr_onnxruntime"):
return "rapidocr", ""
if has_module("pytesseract") and not which("tesseract"):
return None, (
"pytesseract is installed but the tesseract binary isn't on PATH "
"(try: sudo apt install tesseract-ocr)"
)
return None, f"no OCR engine available — {INSTALL_HINT}"
def capture_available(has_module: Callable[[str], bool] = _module_available) -> bool:
return has_module("mss")
def clean_ocr_text(raw: str, max_chars: int = DEFAULT_MAX_CHARS) -> str:
"""Squeeze raw OCR output into something worth sending.
Screen OCR produces a lot of junk single stray glyphs off window
borders, runs of blank lines, the same toolbar label recognised twice. All
of that costs tokens and tells the model nothing, so it goes.
"""
if not raw:
return ""
lines: list[str] = []
for line in raw.splitlines():
line = re.sub(r"[^\S\n]+", " ", line).strip()
if not line:
continue
if len(re.sub(r"[^0-9A-Za-z]", "", line)) < _MIN_MEANINGFUL:
continue
if lines and line == lines[-1]:
continue # consecutive duplicate
lines.append(line)
text = "\n".join(lines)
if max_chars and len(text) > max_chars:
text = text[: max_chars - 1].rstrip() + ""
text += "\n[truncated]"
return text
def format_reading(monitor: Optional[Monitor], text: str) -> str:
"""The tool output handed back for `petctl read`."""
where = f"monitor {monitor.number} ({monitor.name})" if monitor else "screen"
if not text.strip():
return f"[pet] read {where}: no text recognised"
return f"[pet] text on {where}:\n{text}"
# ── capture + recognition (needs a real screen) ──────────────────────────────
def capture(monitor: Monitor):
"""Grab *monitor* as a PIL image, or None if capture isn't available."""
try:
import mss
from PIL import Image
except ImportError:
return None
try:
box = {
"left": monitor.x,
"top": monitor.y,
"width": monitor.width,
"height": monitor.height,
}
with mss.mss() as sct:
shot = sct.grab(box)
return Image.frombytes("RGB", shot.size, shot.bgra, "raw", "BGRX")
except Exception:
return None
def _ocr(image, engine: str) -> str:
if engine == "pytesseract":
import pytesseract
# Grayscale first: tesseract is measurably better on it than on the
# colour desktop, and it's a cheap conversion.
return pytesseract.image_to_string(image.convert("L"))
if engine == "rapidocr":
import numpy as np
from rapidocr_onnxruntime import RapidOCR
result, _ = RapidOCR()(np.array(image))
if not result:
return ""
return "\n".join(line[1] for line in result)
return ""
def read_monitor(monitor: Monitor, max_chars: int = DEFAULT_MAX_CHARS) -> str:
"""OCR one screen and return the formatted tool output.
Never raises: every failure path returns a sentence explaining itself,
because the return value goes straight back to the server as the result of
a command Bolt chose to run.
"""
if not capture_available():
return f"[pet] can't capture the screen — {INSTALL_HINT}"
engine, reason = resolve_engine()
if engine is None:
return f"[pet] can't read the screen — {reason}"
image = capture(monitor)
if image is None:
return (
f"[pet] couldn't capture monitor {monitor.number} "
"(is this a Wayland session? mss needs X11)"
)
try:
raw = _ocr(image, engine)
except Exception as exc:
return f"[pet] OCR failed on monitor {monitor.number}: {exc}"
return format_reading(monitor, clean_ocr_text(raw, max_chars))
def read_monitors(monitors: list[Monitor], max_chars: int = DEFAULT_MAX_CHARS) -> str:
"""OCR several screens, splitting the character budget between them."""
if not monitors:
return "[pet] no monitor information available"
if len(monitors) == 1:
return read_monitor(monitors[0], max_chars)
share = max(400, max_chars // len(monitors))
return "\n\n".join(read_monitor(m, share) for m in monitors)
+235
View File
@@ -0,0 +1,235 @@
"""`petctl self_restart` — the pet restarting itself, and remembering why.
Bolt can already edit this repo through `filectl` and run commands through the
shell relay, which means he can change the pet's own code. What he could not
do is *see the result*: the running process keeps the old modules in memory,
so an edit is invisible until somebody restarts the pet by hand, and by then
the conversation that motivated it is over. That makes the edit-test-review
loop a human errand.
This closes the loop. The tricky part is that the thing being asked to report
back is the thing that dies, so the mechanism is built around three problems:
1. **The turn must survive.** A restart mid-turn would kill the HTTP tool
relay before the result was posted, and the server would sit waiting until
it timed out the conversation lost, with no explanation. So the command
only *arms* the restart: it returns immediately, the turn finishes and Bolt
speaks his reply, and the restart happens after (see
`controller._maybe_self_restart`), exactly like the updater's "only between
turns" rule.
2. **A broken edit must not be fatal.** Before anything is armed, the new code
is imported in a *subprocess* (`preflight`) this process still holds the
old modules, so importing here would prove nothing. A syntax error comes
back as the command's output, in the same turn, and nothing restarts. That
is the difference between "Bolt broke the pet and lost his own way to fix
it" and "Bolt got a traceback and tried again".
3. **The reason must outlive the process.** The context (why, what to check,
which version, when) is written to disk before exec and read on the way
back up, so the new process can open with "I'm back — you asked me to check
X" instead of amnesia. That report goes to the server as a normal turn, so
Bolt sees the result of his own change and can carry on.
A loop guard bounds the worst case: `MAX_RESTARTS` inside `WINDOW_SECONDS`
and further self-restarts are refused with a reason, so an edit-restart-crash
cycle stops on its own rather than spinning the process forever.
Pure-ish and injectable throughout (paths, clock, subprocess runner) so the
whole thing is testable without ever restarting anything.
"""
from __future__ import annotations
import json
import os
import subprocess
import sys
import tempfile
import time
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any, Callable, Optional
from . import config
# Lives in the cache dir, not the repo: it is transient state about *this*
# machine's process, and it must never end up in a git diff of the checkout
# Bolt is editing.
DEFAULT_STATE_PATH = Path.home() / ".cache" / "bolt-pet" / "restart_context.json"
# Loop guard. Deliberately small: a healthy edit-check cycle is one restart
# per change, and anything hammering past this is a crash loop, not work.
MAX_RESTARTS = int(os.environ.get("SELF_RESTART_MAX", "5"))
WINDOW_SECONDS = float(os.environ.get("SELF_RESTART_WINDOW_SECONDS", "900"))
# What the preflight subprocess imports. `ui.app` pulls in the widest slice of
# the package (Qt, controller, audio, every helper), so if this imports, a
# restart will at least reach the event loop.
_PREFLIGHT_IMPORT = "import bolt_pet, bolt_pet.controller, bolt_pet.ui.app"
class RestartError(Exception):
"""A refused restart — reported back to the server as command output."""
@dataclass
class RestartContext:
"""What the dying process wants the next one to know."""
reason: str = ""
verify: str = ""
armed_at: float = 0.0
version: str = ""
session: str = ""
recent: list = field(default_factory=list)
restarts: list = field(default_factory=list) # timestamps, for the loop guard
def as_dict(self) -> dict[str, Any]:
return asdict(self)
def _now() -> float:
return time.time()
def load(path: Optional[Path] = None) -> Optional[RestartContext]:
"""Read the context left by a previous process, or None."""
target = Path(path or DEFAULT_STATE_PATH)
try:
data = json.loads(target.read_text(encoding="utf-8"))
except (FileNotFoundError, json.JSONDecodeError, OSError):
return None
if not isinstance(data, dict):
return None
known = {field_name for field_name in RestartContext().as_dict()}
return RestartContext(**{k: v for k, v in data.items() if k in known})
def save(context: RestartContext, path: Optional[Path] = None) -> None:
"""Persist the context atomically — a half-written file on the way out
would make the next process start confused instead of oriented."""
target = Path(path or DEFAULT_STATE_PATH)
target.parent.mkdir(parents=True, exist_ok=True)
descriptor, temp_path = tempfile.mkstemp(dir=target.parent, prefix=".restart_", suffix=".tmp")
try:
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
json.dump(context.as_dict(), handle, ensure_ascii=False, indent=1)
os.replace(temp_path, target)
except BaseException:
try:
os.unlink(temp_path)
except OSError:
pass
raise
def clear(path: Optional[Path] = None) -> None:
"""Consume the context. Called once it has been reported, so the pet
doesn't announce the same restart every time it starts."""
try:
Path(path or DEFAULT_STATE_PATH).unlink()
except (FileNotFoundError, OSError):
pass
def recent_restarts(context: Optional[RestartContext], *, now: Optional[float] = None) -> list:
current = now if now is not None else _now()
stamps = list((context.restarts if context else []) or [])
return [stamp for stamp in stamps if current - float(stamp) <= WINDOW_SECONDS]
def check_loop_guard(context: Optional[RestartContext], *, now: Optional[float] = None) -> None:
"""Refuse to restart if we've already done it too many times recently."""
stamps = recent_restarts(context, now=now)
if len(stamps) >= MAX_RESTARTS:
raise RestartError(
f"refusing: {len(stamps)} self-restarts in the last "
f"{int(WINDOW_SECONDS / 60)} minutes. Something is looping — fix the "
"cause, or wait for the window to clear before trying again."
)
def preflight(
repo: Optional[Path] = None,
run: Optional[Callable[..., Any]] = None,
timeout: float = 120.0,
) -> None:
"""Import the current source in a subprocess; raise if it's broken.
This process has the *old* modules loaded, so importing in-process would
happily succeed on a file that no longer parses. Mirrors
`updater._smoke_test`, and exists for the same reason: never hand the
session to code that can't start."""
runner = run or subprocess.run
root = Path(repo or config.HERE)
try:
completed = runner(
[sys.executable, "-c", _PREFLIGHT_IMPORT],
cwd=str(root), capture_output=True, text=True, timeout=timeout,
env={**os.environ, "QT_QPA_PLATFORM": "offscreen"}, # no display needed to import
)
except Exception as exc: # subprocess itself failed to run
raise RestartError(f"couldn't run the preflight import check: {exc}") from exc
if completed.returncode != 0:
detail = (completed.stderr or completed.stdout or "").strip()
raise RestartError(
"the current code does not import, so restarting would leave you with "
f"nothing running. Fix this first:\n{detail[-800:]}"
)
def arm(
reason: str,
*,
verify: str = "",
version: str = "",
session: str = "",
recent: Optional[list] = None,
path: Optional[Path] = None,
now: Optional[float] = None,
) -> RestartContext:
"""Record why we're about to die, carrying the restart history forward."""
current = now if now is not None else _now()
previous = load(path)
context = RestartContext(
reason=" ".join(str(reason or "").split())[:400],
verify=" ".join(str(verify or "").split())[:400],
armed_at=current,
version=str(version or ""),
session=str(session or ""),
recent=list(recent or [])[-6:],
restarts=recent_restarts(previous, now=current) + [current],
)
save(context, path)
return context
def report(
context: RestartContext,
*,
version: str = "",
now: Optional[float] = None,
) -> str:
"""The message the new process sends the server on the way up.
Phrased as Bolt reporting to himself, because that is what it is: the
server sees it as an ordinary turn, and the reply comes back through the
normal pipeline which is what lets "restart and check X" finish as a
sentence spoken out loud."""
current = now if now is not None else _now()
took = max(0.0, current - float(context.armed_at or current))
lines = [
"[pet self-restart] I restarted myself and I'm back up.",
f"- reason: {context.reason or 'not recorded'}",
f"- took: {took:.1f}s",
f"- version now running: {version or 'unknown'}"
+ (f" (was {context.version})" if context.version and context.version != version else ""),
]
if context.verify:
lines.append(f"- you wanted to check: {context.verify}")
if context.recent:
lines.append("- what we were doing before: " + " | ".join(str(x)[:120] for x in context.recent))
lines.append(
"The new code is loaded and running. If you wanted to verify something, "
"check it now (filectl to read, command to test) and tell the user what you found."
)
return "\n".join(lines)
+135 -12
View File
@@ -9,26 +9,48 @@ memory, tools, and persona as Discord chat and the Linux voice client:
... -> POST /desk/tool_result (repeat until the server sends a reply)
reply <- returned to caller
A reply can also carry a voice (`voice_id`/`voice_name`), which is how the
server's `speak_as` marker reaches us: Bolt searched the ElevenLabs voice
library, picked one, and tagged the reply with it the client is what
actually speaks in it. See `Reply` and controller._apply_voice.
Kept dependency-free beyond `requests` so it's easy to unit test with mocks.
"""
from __future__ import annotations
import os
import signal
import subprocess
from pathlib import Path
from typing import Callable, Optional
from typing import Callable, NamedTuple, Optional
import requests
from . import config
from . import config, sudo_askpass
_MAX_RELAY_HOPS = 16
# Command output handed back up the relay is capped: it becomes part of the
# server's prompt, and a runaway `find /` would blow the context window.
_MAX_COMMAND_OUTPUT = 6000
class ServerError(Exception):
"""Raised when the server responds with an error payload or unreachable."""
class Reply(NamedTuple):
"""One final reply from the desk API. *voice_id* is set only when the
server tagged this reply with a `speak_as` voice; *voice_name* is the
human-readable name that came with it (may be empty even when the id
isn't). Both empty means "say it in the usual voice"."""
text: str
voice_id: str = ""
voice_name: str = ""
def _headers() -> dict:
return {"X-Desk-Api-Key": config.API_KEY}
@@ -42,26 +64,83 @@ def check_health(timeout: float = 10.0) -> dict:
def run_local_command(command: str, timeout: int = None) -> str:
"""Execute a command relayed by the server, exactly as bolt_desk.py does —
"full desktop control" for things like "open firefox" or "how full is my
disk". Runs as the current desktop user. See README security notes."""
disk". Runs as the current desktop user. See README security notes.
`sudo` gets special handling: the pet has no terminal, so sudo would sit
waiting on a tty that nobody is looking at. With SUDO_ASKPASS_PROMPT on,
bare `sudo` becomes `sudo -A` and the password is collected in a desktop
dialog you have to answer which also gives those commands a longer
timeout, since a human has to notice the window and type."""
env = None
if config.SUDO_ASKPASS_PROMPT and sudo_askpass.needs_password_prompt(command):
helper = sudo_askpass.find_helper()
if helper:
env = sudo_askpass.environment(helper)
command = sudo_askpass.add_askpass_flag(command)
timeout = timeout or config.SUDO_COMMAND_TIMEOUT_SECONDS
timeout = timeout or config.COMMAND_TIMEOUT_SECONDS
try:
completed = subprocess.run(
command, shell=True, capture_output=True, text=True,
timeout=timeout, cwd=str(Path.home()),
# start_new_session puts the shell in its own process group so a timeout
# can kill the whole tree. subprocess.run() would only SIGKILL the `sh`
# itself, leaving whatever it spawned (a build, a `tail -f`, an ffmpeg)
# running forever with no parent watching — one relayed command that
# hangs shouldn't leak a process for the rest of the session.
process = subprocess.Popen(
command, shell=True, cwd=str(Path.home()), env=env, text=True,
stdout=subprocess.PIPE, stderr=subprocess.PIPE,
start_new_session=(os.name == "posix"),
)
output = (completed.stdout or "") + (completed.stderr or "")
return f"[exit {completed.returncode}]\n{output}"[:6000]
except subprocess.TimeoutExpired:
return f"[command timed out after {timeout}s]"
except Exception as exc:
return f"[command failed: {exc}]"
try:
stdout, stderr = process.communicate(timeout=timeout)
return _command_output(f"[exit {process.returncode}]", stdout, stderr)
except subprocess.TimeoutExpired:
stdout, stderr = _terminate(process)
# Whatever it managed to print before it hung is the useful part — a
# bare "timed out" tells the model nothing it can act on, and the last
# line of output usually says exactly what it was stuck waiting for.
return _command_output(f"[command timed out after {timeout}s]", stdout, stderr)
except Exception as exc:
_terminate(process)
return f"[command failed: {exc}]"
def _terminate(process: subprocess.Popen) -> tuple[str, str]:
"""Kill a timed-out command's whole process group and collect what it wrote.
SIGTERM first so a shell script can clean up, SIGKILL a moment later for
anything that ignores it. The final drain is itself time-boxed: a
grandchild holding the pipe open must not turn a timeout into a hang."""
try:
if os.name == "posix":
group = os.getpgid(process.pid)
os.killpg(group, signal.SIGTERM)
try:
process.wait(timeout=2)
except subprocess.TimeoutExpired:
os.killpg(group, signal.SIGKILL)
else:
process.kill()
except (ProcessLookupError, PermissionError, OSError):
pass # already gone, or never had its own group
try:
return process.communicate(timeout=2)
except Exception:
return "", ""
def _command_output(header: str, stdout: Optional[str], stderr: Optional[str]) -> str:
body = (stdout or "") + (stderr or "")
return f"{header}\n{body}"[:_MAX_COMMAND_OUTPUT]
def converse(
text: str,
on_command: Callable[[str], str] = run_local_command,
timeout: float = 120.0,
) -> str:
) -> Reply:
"""Send one turn of conversation to the desk API, relaying any commands
the server sends back until it produces a final reply.
@@ -98,10 +177,54 @@ def converse(
raise ServerError(f"couldn't reach the server during tool relay: {exc}") from exc
if payload.get("type") == "reply":
return str(payload.get("text") or "")
return Reply(
text=str(payload.get("text") or ""),
voice_id=str(payload.get("voice_id") or ""),
voice_name=str(payload.get("voice_name") or ""),
)
if payload.get("type") == "command":
# Fell out of the loop still being handed commands. Worth its own
# message: "unknown server response" sent everyone looking at the
# payload shape, when what actually happened is a model that kept
# calling tools and never answered.
raise ServerError(
f"the server kept relaying commands past the {_MAX_RELAY_HOPS}-hop cap "
"without producing a reply"
)
raise ServerError(str(payload.get("error") or "unknown server response"))
def list_outbox_files(timeout: float = 15.0) -> list:
"""Files the server has queued for this session via its deliver_files
tool (e.g. "send me that report" during a conversation) each entry has
id/name/size. Downloading one (download_outbox_file) dequeues it
server-side, so a file is only ever handed out once."""
try:
response = requests.get(
f"{config.SERVER_URL}/desk/files",
params={"session_id": config.SESSION_ID},
headers=_headers(), timeout=timeout,
)
response.raise_for_status()
return list(response.json().get("files") or [])
except Exception as exc:
raise ServerError(f"couldn't list delivered files: {exc}") from exc
def download_outbox_file(file_id: str, timeout: float = 60.0) -> bytes:
"""Fetches and dequeues one file listed by list_outbox_files()."""
try:
response = requests.get(
f"{config.SERVER_URL}/desk/files/{file_id}",
params={"session_id": config.SESSION_ID},
headers=_headers(), timeout=timeout,
)
response.raise_for_status()
return response.content
except Exception as exc:
raise ServerError(f"couldn't download delivered file {file_id!r}: {exc}") from exc
def report_status(timeout: float = 15.0) -> Optional[str]:
"""Heartbeat — lets the desk API attach a pending spoken announcement
(proactive nudges, reminders fired since the last heartbeat) that the pet
+43
View File
@@ -67,6 +67,27 @@ _SPOKEN_SYMBOLS = {
"=": " equals ",
}
# Abbreviations a voice spells out letter by letter ("eee gee") because the
# periods make them look like sentence boundaries. Written out instead — this
# has to run before _UNSPEAKABLE strips anything, and the trailing \.? keeps
# "etc" working with or without its period. Word-bounded so "vs" inside a
# filename is left alone.
_SPOKEN_ABBREVIATIONS = (
(re.compile(r"\be\.g\.?(?=\s|$)", re.IGNORECASE), "for example"),
(re.compile(r"\bi\.e\.?(?=\s|$)", re.IGNORECASE), "that is"),
(re.compile(r"\betc\.?(?=\s|$)", re.IGNORECASE), "and so on"),
(re.compile(r"\bvs\.?(?=\s|$)", re.IGNORECASE), "versus"),
(re.compile(r"\baka\b", re.IGNORECASE), "also known as"),
(re.compile(r"\bw/(?=\s)", re.IGNORECASE), "with"),
# "PR #42" -> "PR number 42"; a bare "#" is markup and _UNSPEAKABLE drops it.
(re.compile(r"#(?=\d)"), "number "),
# A long option's dashes are punctuation to the eye and syllables to the ear
# ("dash dash force"). Only the doubled form: a single hyphen has to survive
# for "bolt-pet" and "up-to-date", and requiring a word character after it
# keeps a "---" horizontal rule intact for _RULE to strip.
(re.compile(r"(?<!\w)--(?=\w)"), ""),
)
_MULTI_SPACE = re.compile(r"[ \t]+")
_MULTI_PUNCT = re.compile(r"(?:\s*\.){2,}")
@@ -105,6 +126,10 @@ def for_speech(text: str) -> str:
return ""
for symbol, spoken in _PRE_SPOKEN_SYMBOLS.items():
text = text.replace(symbol, spoken)
# Before the markdown pass, so "#42" still has its "#" to word and a real
# "## Heading" (no digit after the hashes) is left for _HEADING to strip.
for pattern, spoken in _SPOKEN_ABBREVIATIONS:
text = pattern.sub(spoken, text)
text = _strip_markdown(text, keep_emoji=False)
text = _URL.sub(" link ", text)
text = _TABLE_PIPE.sub(", ", text)
@@ -118,6 +143,24 @@ def for_speech(text: str) -> str:
return text.strip()
def is_question(text: str) -> bool:
"""True if the reply asks the user anything — the cue for the pet to keep
listening instead of making you say the wake word again.
Anywhere in the reply counts, not only the end. An earlier version required
a *trailing* '?' on the theory that "What time is it? It's 7:15." isn't
waiting on an answer, and that's true of that sentence but wrong far more
often: Bolt routinely asks first and then keeps talking ("Want me to fix
it? I'd start with the config."), and refusing to listen there is the case
that actually costs you a wake word. The cheap failure is the other
direction an unwanted extra listen ends itself on `VAD_GRACE_SECONDS` of
silence, and `FOLLOW_UP_MAX_TURNS` caps the chain.
The test runs on the *spoken* form, so a '?' that only exists inside a
stripped code block, a URL, or a markdown link target doesn't count."""
return "?" in for_speech(text)
def for_display(text: str) -> str:
"""What the speech bubble shows: markdown syntax removed (the bubble
can't render it) but emoji and layout-ish punctuation left alone."""
+6 -1
View File
@@ -26,11 +26,16 @@ class PetState(str, Enum):
# make the pet speak unprompted — a reminder firing, a nudge from the server
# — without the user having said anything first, so there's no preceding
# LISTENING/THINKING leg for that turn.
#
# TALKING -> THINKING is the mirror case: a `dialoguectl` scene is played
# *mid-turn*, while the server is still waiting on the tool result, so the pet
# talks and then goes back to waiting rather than falling to IDLE (which would
# make it look like the turn had ended).
_TRANSITIONS: dict[PetState, set[PetState]] = {
PetState.IDLE: {PetState.LISTENING, PetState.TALKING, PetState.ERROR},
PetState.LISTENING: {PetState.THINKING, PetState.IDLE, PetState.ERROR},
PetState.THINKING: {PetState.TALKING, PetState.IDLE, PetState.ERROR},
PetState.TALKING: {PetState.IDLE, PetState.ERROR},
PetState.TALKING: {PetState.IDLE, PetState.THINKING, PetState.ERROR},
PetState.ERROR: {PetState.IDLE},
}
+145
View File
@@ -0,0 +1,145 @@
"""Graphical password prompts for server-relayed `sudo` commands.
The pet has no terminal. When the server relays something like
`sudo apt update`, sudo tries to read a password from a tty, finds none (or
finds the terminal the pet was launched from, which you're not looking at),
and the command fails with no way to answer it.
sudo's own answer to this is SUDO_ASKPASS: with `-A`, it runs a helper
program and reads the password from the helper's stdout instead of a tty.
Any GUI prompt that prints what was typed works, so this module finds a real
askpass binary if one is installed and otherwise generates a one-line wrapper
around zenity/kdialog, which every desktop has one of.
Worth being clear about what this changes: the prompt is a *feature*, not
just plumbing. Server-relayed commands already run as your desktop user (see
server_client.run_local_command); this lets them ask to run as root, and the
dialog is the only thing standing between "Bolt decided to run sudo" and it
happening. Leave SUDO_ASKPASS_PROMPT on, and read the dialogs.
The parts that decide *what* to run are pure functions so they're tested
without a display, a password, or a working sudo.
"""
from __future__ import annotations
import os
import re
import shutil
import stat
from pathlib import Path
from typing import Callable, Optional
from . import config
# Real askpass binaries, in preference order. These are purpose-built for
# this (they grab the keyboard, hide the input, and don't leave the password
# in a process argument), so they win over a generated wrapper.
_KNOWN_HELPERS = (
"/usr/bin/ssh-askpass",
"/usr/lib/ssh/ssh-askpass",
"/usr/lib/openssh/gnome-ssh-askpass3",
"/usr/lib/openssh/gnome-ssh-askpass",
"/usr/libexec/openssh/ssh-askpass",
"/usr/bin/ksshaskpass",
"/usr/bin/lxqt-openssh-askpass",
)
# Dialog tools we can wrap when no askpass binary exists. Each must print the
# typed password to stdout and nothing else.
_WRAPPABLE = {
"zenity": '{tool} --password --title="Bolt" --text="$1" 2>/dev/null',
"kdialog": '{tool} --password "$1" --title "Bolt" 2>/dev/null',
}
# `sudo` at the start of the command or right after a shell separator, not
# already carrying a flag. Deliberately conservative: a `sudo` inside a
# quoted string or a heredoc is left alone, because rewriting it could change
# what the command means.
_SUDO = re.compile(r"(^|[;&|]\s*|\n\s*)(sudo)(\s+)(?!-)")
def add_askpass_flag(command: str) -> str:
"""Insert `-A` after each bare `sudo`, so it prompts through the helper
instead of a tty. Commands that already pass a flag (`sudo -n`, `sudo -A`,
`sudo -u bob`) are left exactly as they are the caller was explicit."""
return _SUDO.sub(r"\1\2 -A\3", command or "")
def needs_password_prompt(command: str) -> bool:
"""True if *command* has a `sudo` that might sit waiting on a dialog.
Used to give those commands a longer timeout 30 seconds is fine for a
shell command and nowhere near enough for a human to notice a window,
read it, and type a password."""
return bool(_SUDO.search(command or ""))
def helper_script(tool_path: str) -> str:
"""The wrapper script for a dialog *tool_path*. sudo passes its prompt
("[sudo] password for maji:") as $1, which is worth showing it names
the user the password is for."""
name = Path(tool_path).name
body = _WRAPPABLE[name].format(tool=tool_path)
return f"#!/bin/sh\n# Generated by bolt-pet. Prints the typed password on stdout for sudo -A.\n{body}\n"
def _default_cache_dir() -> Path:
base = os.environ.get("XDG_CACHE_HOME") or (Path.home() / ".cache")
return Path(base) / "bolt-pet"
def find_helper(
configured: str = None,
is_executable: Callable[[str], bool] = None,
which: Callable[[str], Optional[str]] = None,
cache_dir: Path = None,
write: bool = True,
) -> Optional[str]:
"""Path to an askpass helper, or None if the desktop has nothing we can
use. Order: whatever SUDO_ASKPASS_HELPER names, then a real askpass
binary, then a generated wrapper around zenity/kdialog.
The lookups are injectable so the resolution order is testable on a box
with a different set of these installed than yours."""
configured = config.SUDO_ASKPASS_HELPER if configured is None else configured
is_executable = is_executable or (lambda path: os.path.isfile(path) and os.access(path, os.X_OK))
which = which or shutil.which
if configured:
return configured if is_executable(configured) else None
for candidate in _KNOWN_HELPERS:
if is_executable(candidate):
return candidate
for tool in _WRAPPABLE:
tool_path = which(tool)
if not tool_path:
continue
if not write:
return tool_path
return _write_wrapper(tool_path, cache_dir or _default_cache_dir())
return None
def _write_wrapper(tool_path: str, cache_dir: Path) -> Optional[str]:
"""Drop the wrapper script somewhere sudo can execute it. Mode 0700: it
isn't secret, but it's a thing that pops up a password box, so nobody
else on the machine gets to edit it."""
try:
cache_dir.mkdir(parents=True, exist_ok=True)
script = cache_dir / "askpass.sh"
source = helper_script(tool_path)
if not script.exists() or script.read_text(encoding="utf-8") != source:
script.write_text(source, encoding="utf-8")
script.chmod(stat.S_IRWXU)
return str(script)
except Exception:
return None # no prompt is better than a crashed command relay
def environment(helper: str, base: dict = None) -> dict:
"""The subprocess environment with SUDO_ASKPASS pointed at *helper*."""
env = dict(os.environ if base is None else base)
env["SUDO_ASKPASS"] = helper
return env
+34 -2
View File
@@ -13,7 +13,7 @@ import sys
from PySide6.QtCore import QThread
from PySide6.QtWidgets import QApplication
from .. import config
from .. import config, updater
from ..controller import PetController
from ..hotkey import GlobalHotkey
from ..state import PetState
@@ -46,6 +46,13 @@ def run() -> int:
controller.finished.connect(thread.quit)
window.talk_requested.connect(controller.request_talk_now)
# The window owns the screen list and tells the controller about it, so
# both ends agree on what "monitor 2" means (see monitors.py).
window.monitors_changed.connect(controller.set_monitors)
window.pet_monitor_changed.connect(controller.set_pet_monitor)
# PetWindow publishes once in its constructor, which ran before those
# connections existed — so say it again now that anyone is listening.
window.publish_monitors()
window.copied.connect(lambda text: _log(f"Copied to clipboard: {text[:60]}"))
history_window = HistoryWindow(controller.history)
@@ -73,6 +80,7 @@ def run() -> int:
on_set_nap=_set_nap,
on_show_history=history_window.show_refreshed,
on_show_wake_tuner=tuner_window.show_refreshed,
on_reset_voice=controller.reset_voice,
)
def _handle_napping(napping: bool) -> None:
@@ -80,6 +88,9 @@ def run() -> int:
tray.set_napping(napping)
controller.napping.connect(_handle_napping)
# The server can hand Bolt a different voice mid-conversation (speak_as);
# the tray is where you get his own back.
controller.voice_changed.connect(tray.set_voice)
# Push-to-talk: a global hook, because the pet window never has focus.
# request_talk_now() only sets a threading.Event, so it's safe to call
@@ -91,6 +102,19 @@ def run() -> int:
elif hotkey.running:
_log(f"Push-to-talk: {config.PUSH_TO_TALK_HOTKEY}")
# The updater has already moved the checkout by the time this fires; all
# that's left is to let Qt tear down cleanly (so the mic and the tray
# icon are released) and then exec the new code. Doing the exec after
# app.exec() returns, rather than from the controller thread, is what
# guarantees the audio device is free before the new process opens it.
pending_restart = {"tag": None}
def _handle_restart(tag: str) -> None:
pending_restart["tag"] = tag
app.quit()
controller.restart_requested.connect(_handle_restart)
def _shutdown() -> None:
hotkey.stop()
controller.stop()
@@ -100,4 +124,12 @@ def run() -> int:
app.aboutToQuit.connect(_shutdown)
thread.start()
return app.exec()
status = app.exec()
if pending_restart["tag"]:
_log(f"Restarting into {pending_restart['tag']}")
try:
updater.restart() # never returns
except Exception as exc:
_log(f"Couldn't restart automatically ({exc}) — start the pet again by hand.")
return status
+188 -5
View File
@@ -20,8 +20,9 @@ from PySide6.QtGui import (
from PySide6.QtWidgets import QApplication, QWidget
from .. import config
from ..monitors import Monitor
from ..state import PetState
from .sprite import SpriteSet
from .sprite import WALK, SpriteSet
_DRAG_THRESHOLD_PX = 4
# Movement runs on its own ~30fps timer, independent of the (slower) sprite
@@ -29,6 +30,12 @@ _DRAG_THRESHOLD_PX = 4
_WANDER_TICK_MS = 33
_EMOTE_TICKS = 36 # ~1.2s per emote at the tick rate above
_NAP_OPACITY = 0.35
# How far the pet travels per walk-cycle frame. The cycle is advanced by
# distance rather than by the animation clock so a planted paw tracks backwards
# at exactly the speed the window moves forwards — drive it off a timer instead
# and the feet skate whenever PET_WANDER_SPEED doesn't happen to match the fps.
# Eight frames at 13px is a ~104px stride cycle, a bit under the pet's width.
_WALK_PIXELS_PER_FRAME = 13.0
def emote_transform(emote: str, progress: float) -> tuple[float, float, float, float]:
@@ -164,6 +171,12 @@ class SpeechBubble(QWidget):
class PetWindow(QWidget):
talk_requested = Signal()
copied = Signal(str) # bubble text the user just put on the clipboard
# The screen layout, published *to* the controller (queued, cross-thread).
# The window is the only thing allowed to ask Qt about screens, so the
# controller and the window can never disagree about what "monitor 2"
# means — see monitors.py.
monitors_changed = Signal(list) # list[monitors.Monitor]
pet_monitor_changed = Signal(int) # 0-based index the pet is standing on
def __init__(self, sprite_dir: Optional[Path] = None, size: Optional[int] = None):
super().__init__()
@@ -201,6 +214,10 @@ class PetWindow(QWidget):
self._next_wander_at = 0.0
self._bob_offset = 0
self._bob_phase = 0.0
self._walking = False
self._facing = 1 # +1 right, -1 left; the walk art is drawn facing right
self._walk_distance = 0.0
self._mirror_cache: dict[int, QPixmap] = {}
self._schedule_next_wander()
self._wander_timer = QTimer(self)
self._wander_timer.timeout.connect(self._movement_tick)
@@ -210,6 +227,18 @@ class PetWindow(QWidget):
self.set_click_through(config.PET_CLICK_THROUGH)
self._place_start_position()
self._monitors: list[Monitor] = []
self._pet_monitor: Optional[int] = None
self._last_published_pos: Optional[QPoint] = None
app = QApplication.instance()
if app is not None:
# Screens come and go — a laptop docking, a TV waking up. Republish
# rather than letting Bolt jump to a monitor that's been unplugged.
app.screenAdded.connect(lambda _s: self.publish_monitors())
app.screenRemoved.connect(lambda _s: self.publish_monitors())
app.primaryScreenChanged.connect(lambda _s: self.publish_monitors())
self.publish_monitors()
# ── placement ────────────────────────────────────────────────────────
def _place_start_position(self) -> None:
@@ -244,6 +273,8 @@ class PetWindow(QWidget):
if target is not None:
self._wander_target = target
self._commanded_move = True # overrides the idle-only rule
elif kind == "jump":
self.jump_to_monitor(int(action["monitor"]))
elif kind == "emote":
self.start_emote(action["emote"])
elif kind == "say":
@@ -378,6 +409,98 @@ class PetWindow(QWidget):
"""Stroll immediately (tray menu / anything that wants a nudge)."""
self._next_wander_at = 0.0
# ── monitors ─────────────────────────────────────────────────────────
def _build_monitors(self) -> list[Monitor]:
"""Snapshot Qt's screen list as plain dataclasses.
Full `geometry()`, not `availableGeometry()`: these coordinates are
what a screen grab gets cropped to, and a grab doesn't stop at the
taskbar. Placement uses availableGeometry separately.
"""
primary = QApplication.primaryScreen()
out = []
for index, screen in enumerate(QApplication.screens()):
geo = screen.geometry()
out.append(
Monitor(
index=index,
name=screen.name() or f"screen-{index + 1}",
x=geo.x(),
y=geo.y(),
width=geo.width(),
height=geo.height(),
primary=screen is primary,
)
)
return out
def publish_monitors(self, force: bool = True) -> None:
"""Push the current layout to whoever's listening (the controller).
*force* re-emits even when nothing changed, which is what the initial
wiring in ui/app.py needs: this window is built before the controller
exists, so the constructor's first publish goes to nobody.
"""
monitors = self._build_monitors()
changed = monitors != self._monitors
self._monitors = monitors
if changed or force:
self.monitors_changed.emit(monitors)
self._publish_pet_monitor(force=True)
def monitors(self) -> list[Monitor]:
return list(self._monitors)
def current_monitor_index(self) -> Optional[int]:
center = self.frameGeometry().center()
screens = QApplication.screens()
if not screens:
return None
screen = QApplication.screenAt(center)
if screen is not None:
try:
return screens.index(screen)
except ValueError:
pass
# Straddling a gap or dragged off the desktop entirely — fall back to
# whichever screen centre is nearest rather than reporting nothing.
best = min(
range(len(screens)),
key=lambda i: (screens[i].geometry().center() - center).manhattanLength(),
)
return best
def _publish_pet_monitor(self, force: bool = False) -> None:
index = self.current_monitor_index()
if index is None:
return
if force or index != self._pet_monitor:
self._pet_monitor = index
self.pet_monitor_changed.emit(index)
def jump_to_monitor(self, index: int) -> None:
"""Teleport to *index* (0-based, resolved by the controller) and land
with a hop. Instant rather than a stroll Bolt asked to *jump*, and
walking between screens would take the long way across the desktop."""
screens = QApplication.screens()
if not (0 <= index < len(screens)):
return
geo = screens[index].availableGeometry()
point = self._clamp_to_screen(
QPoint(
geo.left() + (geo.width() - self.width()) // 2,
geo.top() + (geo.height() - self.height()) // 2,
),
geo,
)
self._stop_walking() # drop any stroll in flight, or it walks straight back
self.move(point)
self._schedule_next_wander()
self._publish_pet_monitor(force=True)
self.start_emote("hop")
self.update()
def _screen_geometry(self):
# screenAt() so a multi-monitor setup keeps the pet on the screen
# it's currently standing on rather than yanking it to the primary.
@@ -389,12 +512,17 @@ class PetWindow(QWidget):
self._next_wander_at = time.monotonic() + random.uniform(0.5 * base, 1.5 * base)
def _stop_walking(self) -> None:
if self._wander_target is None and not self._bob_offset:
if self._wander_target is None and not self._bob_offset and not self._walking:
return
self._wander_target = None
self._commanded_move = False
self._bob_phase = 0.0
self._bob_offset = 0
self._walking = False
self._walk_distance = 0.0
# Back to a standing frame, so the next stroll starts from a contact
# pose instead of mid-stride.
self.sprites.get(WALK).reset()
self.update()
def snap_to_edge(self) -> bool:
@@ -447,6 +575,13 @@ class PetWindow(QWidget):
wandering only happens when it's otherwise unoccupied."""
self._advance_emote()
self._wander_tick()
# Report crossing a screen boundary — by strolling, by being dragged,
# by anything. Guarded on the position actually changing so the common
# case (a stationary pet, 30x a second) costs one comparison.
position = self.pos()
if position != self._last_published_pos:
self._last_published_pos = position
self._publish_pet_monitor()
def _wander_tick(self) -> None:
# Only stroll while genuinely idle: not mid-drag, not napping, not
@@ -484,11 +619,51 @@ class PetWindow(QWidget):
self.snap_to_edge()
else:
self.move(round(here.x() + dx / distance * step), round(here.y() + dy / distance * step))
self._bob_phase += 0.45 # little walk-cycle hop
self._bob_offset = int(round(-2.5 * abs(math.sin(self._bob_phase))))
self._advance_walk(dx, dy, step)
self.update()
self._reposition_bubble()
def _advance_walk(self, dx: float, dy: float, step: float) -> None:
"""Drive the walk cycle from distance travelled (see the constant).
Falls back to the old bob-in-code if there's no walk art, so a sprite
folder without a walk/ directory still looks like it's moving rather
than sliding perfectly flat.
"""
self._walking = True
# Only turn on meaningful horizontal travel: a near-vertical stroll
# would otherwise flip him back and forth on rounding noise.
if abs(dx) > 1.0:
self._facing = 1 if dx > 0 else -1
if not self.sprites.has(WALK):
self._bob_phase += 0.45
self._bob_offset = int(round(-2.5 * abs(math.sin(self._bob_phase))))
return
self._bob_offset = 0 # the walk frames carry their own weight shift
self._walk_distance += step
while self._walk_distance >= _WALK_PIXELS_PER_FRAME:
self._walk_distance -= _WALK_PIXELS_PER_FRAME
self.sprites.get(WALK).advance()
def _animation_key(self):
"""Walking overrides the state animation — but only while genuinely
idle-and-moving, so he doesn't trot on the spot mid-sentence."""
if self._walking and self.sprites.has(WALK):
return WALK
return self._current_state
def _oriented(self, pixmap: Optional[QPixmap]) -> Optional[QPixmap]:
"""Mirror the (right-facing) walk art when he's heading left. Cached
per source frame flipping on every paint would be wasteful at 30fps."""
if pixmap is None or self._facing >= 0:
return pixmap
key = pixmap.cacheKey()
mirrored = self._mirror_cache.get(key)
if mirrored is None:
mirrored = pixmap.transformed(QTransform().scale(-1, 1), Qt.SmoothTransformation)
self._mirror_cache[key] = mirrored
return mirrored
# ── state / speech ──────────────────────────────────────────────────
def set_state(self, state: PetState) -> None:
@@ -514,6 +689,11 @@ class PetWindow(QWidget):
# ── animation ────────────────────────────────────────────────────────
def _advance_frame(self) -> None:
# While walking the cycle is stepped by _advance_walk from distance
# travelled; letting this timer also advance it would double-step it
# and put the feet out of sync with the movement.
if self._walking and self.sprites.has(WALK):
return
self.sprites.get(self._current_state).advance()
self.update()
@@ -521,7 +701,10 @@ class PetWindow(QWidget):
painter = QPainter(self)
painter.setRenderHint(QPainter.Antialiasing)
painter.setRenderHint(QPainter.SmoothPixmapTransform)
pixmap: Optional[QPixmap] = self.sprites.get(self._current_state).current()
key = self._animation_key()
pixmap: Optional[QPixmap] = self.sprites.get(key).current()
if key == WALK:
pixmap = self._oriented(pixmap)
if pixmap is None:
self._apply_input_mask(None, 0, 0)
return
+35 -6
View File
@@ -95,17 +95,46 @@ def _load_frames_from_dir(directory: Path, size: int) -> list[QPixmap]:
return frames
WALK = "walk"
# Animations that aren't pipeline states. Walking is a property of *movement*,
# orthogonal to whether the pet is idle/listening/talking, so it deliberately
# isn't a PetState — state.py stays a description of the conversation, not of
# the body. Loaded the same way, keyed by name.
EXTRA_ANIMATIONS = (WALK,)
class SpriteSet:
"""All animations for every PetState, loaded from *sprite_dir*."""
"""All animations for every PetState, plus the extras, from *sprite_dir*."""
def __init__(self, sprite_dir: Path = DEFAULT_SPRITE_DIR, size: int = 160):
self.size = size
self._animations: dict[PetState, SpriteAnimation] = {}
self._animations: dict[str, SpriteAnimation] = {}
self._loaded: set[str] = set() # keys backed by real art, not placeholders
for state in PetState:
frames = _load_frames_from_dir(sprite_dir / state.value, size)
if not frames:
if frames:
self._loaded.add(state.value)
else:
frames = _placeholder_frames(state, size)
self._animations[state] = SpriteAnimation(frames)
self._animations[state.value] = SpriteAnimation(frames)
for name in EXTRA_ANIMATIONS:
frames = _load_frames_from_dir(sprite_dir / name, size)
if frames:
self._loaded.add(name)
self._animations[name] = SpriteAnimation(frames)
def get(self, state: PetState) -> SpriteAnimation:
return self._animations[state]
@staticmethod
def _key(key) -> str:
return key.value if isinstance(key, PetState) else str(key)
def get(self, key) -> SpriteAnimation:
"""Animation for a PetState or an extra name. Unknown/absent extras
fall back to idle, so a sprite folder with no walk/ still runs."""
return self._animations.get(self._key(key)) or self._animations[PetState.IDLE.value]
def has(self, key) -> bool:
"""True only when real frames were found — the caller uses this to
decide whether to use an extra animation at all, rather than being
handed a placeholder blob that looks nothing like walking."""
return self._key(key) in self._loaded
+23 -1
View File
@@ -1,6 +1,7 @@
"""System tray icon — the pet window is frameless with no taskbar entry, so
this menu is the only always-available way to control or exit it: talk now,
mute, wander, click-through, nap, history, wake-word tuning, quit.
mute, wander, click-through, nap, history, wake-word tuning, voice reset,
quit.
Every entry is a plain callback passed in by ui/app.py; this file knows
nothing about the controller or the pet window.
@@ -46,6 +47,7 @@ class PetTray(QSystemTrayIcon):
on_set_nap: Optional[Callable[[bool], None]] = None,
on_show_history: Optional[Callable[[], None]] = None,
on_show_wake_tuner: Optional[Callable[[], None]] = None,
on_reset_voice: Optional[Callable[[], None]] = None,
parent=None,
):
super().__init__(_make_icon(muted=False), parent)
@@ -102,6 +104,16 @@ class PetTray(QSystemTrayIcon):
tuner_action.triggered.connect(on_show_wake_tuner)
menu.addAction(tuner_action)
# Only ever enabled while a server-picked voice (speak_as) is in use —
# it's the way back from "talk like a pirate", which nothing else
# undoes short of a restart.
self._voice_action = None
if on_reset_voice is not None:
self._voice_action = QAction("Use default voice", menu)
self._voice_action.setEnabled(False)
self._voice_action.triggered.connect(on_reset_voice)
menu.addAction(self._voice_action)
menu.addSeparator()
quit_action = QAction("Quit", menu)
quit_action.triggered.connect(on_quit)
@@ -115,6 +127,16 @@ class PetTray(QSystemTrayIcon):
self._mute_action.setChecked(self._muted)
self._refresh_icon()
def set_voice(self, voice: str) -> None:
"""Reflect the voice the controller is speaking in — a name (or id)
when the server picked one, "" for Bolt's own."""
if self._voice_action is None:
return
self._voice_action.setEnabled(bool(voice))
self._voice_action.setText(
f"Use default voice (now: {voice})" if voice else "Use default voice"
)
def set_napping(self, napping: bool) -> None:
"""Reflect a nap the *controller* decided on (quiet hours, fullscreen,
or a petctl command) not just ones clicked here."""
+323
View File
@@ -0,0 +1,323 @@
"""Self-update from the Gitea releases page.
Polls `<UPDATE_REPO_API>/releases/latest` for a tag newer than
``bolt_pet.__version__`` and, if there is one, moves the checkout to that tag
and restarts the pet. The install is expected to be a git clone (which is how
it's deployed), so "download the update" is just `git fetch` + `git checkout`
atomic, and the previous ref is one command away if anything goes wrong.
Safety rules, in the order they're enforced:
1. **A dirty working tree is never touched.** Local edits are skipped over,
not stashed the pet silently discarding your work-in-progress would be
far worse than running an old version.
2. **Everything after checkout is guarded.** Dependency install and an import
smoke test both run before the restart; if either fails, the checkout is
rolled back to the exact ref that was live before (branch name if we were
on one, otherwise the commit) and the deps reinstalled from it.
3. **The restart only happens once the new code imports.** So a broken
release costs you a rollback and a log line, not a pet that won't start.
The git side goes through an injectable *run* callable ``(args) ->
(returncode, output)`` so the whole apply/rollback dance is unit-tested
against a fake git rather than a real repo. Version comparison and release
parsing are pure functions for the same reason.
"""
from __future__ import annotations
import os
import subprocess
import sys
from dataclasses import dataclass
from pathlib import Path
from typing import Callable, Optional, Tuple
import requests
from . import config
# (returncode, combined stdout+stderr)
GitResult = Tuple[int, str]
GitRunner = Callable[[list], GitResult]
_GIT_TIMEOUT_SECONDS = 300
class UpdateError(Exception):
"""Raised when an update can't be applied. If it's raised *after* the
checkout moved, the rollback has already run."""
@dataclass(frozen=True)
class Release:
tag: str
name: str
body: str
prerelease: bool
# ── pure logic ───────────────────────────────────────────────────────────────
def parse_version(tag: str) -> tuple:
"""``"v1.2.3"`` -> ``(1, 2, 3)``. Leading "v" optional; a trailing
suffix ends the parse (``"1.2.3-beta1"`` -> ``(1, 2, 3)``), so a
prerelease of a version compares equal to it rather than sorting
randomly. Junk parses to ``()``, which is never newer than anything."""
parts: list[int] = []
for chunk in (tag or "").strip().lstrip("vV").split("."):
digits = ""
for char in chunk:
if not char.isdigit():
break
digits += char
if not digits:
break
parts.append(int(digits))
return tuple(parts)
def is_newer(candidate: str, current: str) -> bool:
"""True if *candidate* is a strictly newer version than *current*.
Compares zero-padded, so 1.2 == 1.2.0 and 1.2.1 > 1.2."""
new, old = parse_version(candidate), parse_version(current)
if not new:
return False
width = max(len(new), len(old))
return new + (0,) * (width - len(new)) > old + (0,) * (width - len(old))
def release_from_payload(payload: dict) -> Optional[Release]:
"""Gitea's release JSON -> Release, or None if it's a draft or has no
tag. /releases/latest already excludes drafts and prereleases, but the
same parser is used for the full list."""
if not isinstance(payload, dict) or payload.get("draft"):
return None
tag = str(payload.get("tag_name") or "").strip()
if not tag:
return None
return Release(
tag=tag,
name=str(payload.get("name") or tag),
body=str(payload.get("body") or ""),
prerelease=bool(payload.get("prerelease")),
)
# ── talking to Gitea ─────────────────────────────────────────────────────────
def fetch_latest_release(api_url: str = None, token: str = None, timeout: float = 15.0) -> Optional[Release]:
"""Newest published release, or None if the repo has no releases yet
(a fresh repo 404s here, which is not an error worth logging every hour).
Raises UpdateError if the server is unreachable or answers with junk."""
api_url = (api_url if api_url is not None else config.UPDATE_REPO_API).rstrip("/")
if not api_url:
raise UpdateError("UPDATE_REPO_API is not set")
token = config.UPDATE_TOKEN if token is None else token
headers = {"Authorization": f"token {token}"} if token else {}
try:
response = requests.get(f"{api_url}/releases/latest", headers=headers, timeout=timeout)
except Exception as exc:
raise UpdateError(f"couldn't reach the releases API: {exc}") from exc
if response.status_code == 404:
return None
try:
response.raise_for_status()
payload = response.json()
except Exception as exc:
raise UpdateError(f"bad response from the releases API: {exc}") from exc
return release_from_payload(payload)
def check_for_update(current_version: str = None, **kwargs) -> Optional[Release]:
"""The whole "is there anything new?" question in one call. Returns the
Release to move to, or None if we're already current."""
from . import __version__
current = __version__ if current_version is None else current_version
release = fetch_latest_release(**kwargs)
if release is None or not is_newer(release.tag, current):
return None
return release
# ── git ──────────────────────────────────────────────────────────────────────
def git_runner(repo: Path = None) -> GitRunner:
repo = Path(repo or config.HERE)
def run(args: list) -> GitResult:
try:
completed = subprocess.run(
["git", *args], cwd=str(repo), capture_output=True,
text=True, timeout=_GIT_TIMEOUT_SECONDS,
)
except Exception as exc:
return 1, f"git {' '.join(args)} failed to start: {exc}"
return completed.returncode, ((completed.stdout or "") + (completed.stderr or "")).strip()
return run
def is_git_clone(run: GitRunner) -> bool:
return run(["rev-parse", "--git-dir"])[0] == 0
def working_tree_dirty(run: GitRunner) -> bool:
code, output = run(["status", "--porcelain"])
return code != 0 or bool(output.strip())
def current_ref(run: GitRunner) -> str:
"""The branch name if we're on one, else the commit SHA — i.e. whatever
`git checkout` needs to put things back exactly as they were."""
code, output = run(["symbolic-ref", "--quiet", "--short", "HEAD"])
if code == 0 and output.strip():
return output.strip()
code, output = run(["rev-parse", "HEAD"])
if code != 0 or not output.strip():
raise UpdateError("couldn't work out the current git ref")
return output.strip()
def already_at_tag(run: GitRunner, tag: str) -> bool:
"""True if HEAD is already the commit *tag* points at.
Guards the loop you get when a release is cut without bumping
__version__ in the tagged commit: the checkout succeeds (it's a no-op),
the pet restarts, reads the same old __version__, sees the same "newer"
tag, and does it again a restart every check, forever."""
code, head = run(["rev-parse", "HEAD"])
if code != 0 or not head.strip():
return False
code, target = run(["rev-parse", f"tags/{tag}^{{commit}}"])
if code != 0 or not target.strip():
return False # tag isn't known locally yet, so we're certainly not on it
return head.strip() == target.strip()
def _requirements_changed(run: GitRunner, before: str, after: str) -> bool:
code, output = run(["diff", "--name-only", before, after, "--", "requirements.txt"])
return code == 0 and bool(output.strip())
def _install_deps(repo: Path) -> None:
completed = subprocess.run(
[sys.executable, "-m", "pip", "install", "-r", "requirements.txt"],
cwd=str(repo), capture_output=True, text=True, timeout=_GIT_TIMEOUT_SECONDS,
)
if completed.returncode != 0:
raise UpdateError(f"pip install failed: {(completed.stderr or '')[-500:]}")
def _smoke_test(repo: Path) -> None:
"""Import the freshly checked-out package in a *subprocess* — this one
still has the old modules loaded, so importing here would prove nothing.
Catches the common broken release (syntax error, missing dependency)
before we hand the session over to it."""
completed = subprocess.run(
[sys.executable, "-c", "import bolt_pet; import bolt_pet.controller"],
cwd=str(repo), capture_output=True, text=True, timeout=120,
)
if completed.returncode != 0:
raise UpdateError(f"the new version failed to import: {(completed.stderr or '')[-500:]}")
def apply_update(
tag: str,
run: GitRunner = None,
repo: Path = None,
on_log: Callable[[str], None] = lambda _msg: None,
install_deps: bool = None,
verify: Callable[[Path], None] = None,
) -> str:
"""Move the checkout to *tag*, rolling back to where it was if anything
downstream of the checkout fails. Returns the ref we came from (handy for
logging / a manual `git checkout` back). Raises UpdateError otherwise."""
repo = Path(repo or config.HERE)
run = run or git_runner(repo)
install_deps = config.UPDATE_INSTALL_DEPS if install_deps is None else install_deps
verify = _smoke_test if verify is None else verify
if not is_git_clone(run):
raise UpdateError("not a git clone — auto-update only works on a git checkout")
if working_tree_dirty(run):
raise UpdateError("working tree has local changes — skipping (nothing was touched)")
previous = current_ref(run)
code, output = run(["fetch", "--tags", "--prune", config.UPDATE_GIT_REMOTE])
if code != 0:
raise UpdateError(f"git fetch failed: {output}")
if already_at_tag(run, tag):
from . import __version__
raise UpdateError(
f"already checked out {tag}, but __version__ still reads {__version__}"
f"bump it in the tagged commit, or every check re-applies the same release"
)
code, output = run(["checkout", "--force", f"tags/{tag}"])
if code != 0:
raise UpdateError(f"git checkout {tag} failed: {output}")
on_log(f"Checked out {tag} (was {previous}).")
# Past this point every failure has to put the checkout back.
try:
if install_deps and _requirements_changed(run, previous, f"tags/{tag}"):
on_log("requirements.txt changed — installing.")
_install_deps(repo)
verify(repo)
except UpdateError as exc:
_rollback(run, previous, repo, on_log, install_deps)
raise UpdateError(f"{exc} — rolled back to {previous}") from exc
except Exception as exc: # a verify() that blows up is still a failed update
_rollback(run, previous, repo, on_log, install_deps)
raise UpdateError(f"update failed ({exc}) — rolled back to {previous}") from exc
return previous
def _rollback(
run: GitRunner,
previous: str,
repo: Path,
on_log: Callable[[str], None],
install_deps: bool,
) -> None:
"""Best-effort return to *previous*. Never raises — it's already running
inside a failure path, and the caller's UpdateError is the thing worth
surfacing. A rollback that itself fails gets its own loud log line,
because that's the one case needing a human."""
code, output = run(["checkout", "--force", previous])
if code != 0:
on_log(f"ROLLBACK FAILED — the checkout is stranded. Run: git checkout {previous} ({output})")
return
on_log(f"Rolled back to {previous}.")
if install_deps:
try:
_install_deps(repo)
except Exception as exc:
on_log(f"Rolled back, but reinstalling the old requirements failed: {exc}")
# ── restart ──────────────────────────────────────────────────────────────────
def restart() -> None:
"""Replace this process with a fresh `python -m bolt_pet`.
execv rather than spawn-and-exit so there's no window with two pets
holding the same mic, and no orphan if the parent dies first. Never
returns when it works; callers should have shut the Qt app and released
the audio device before calling it.
chdir first because `-m bolt_pet` resolves the package from the working
directory: the pet may well have been launched from somewhere else
(autostart entry, run.sh invoked by path), and the new process has to
land on the checkout the update was just applied to."""
os.chdir(str(config.HERE))
os.execv(sys.executable, [sys.executable, "-m", "bolt_pet"])
+12
View File
@@ -32,6 +32,18 @@ pynput>=1.7
# Not imported by the app itself.
Pillow>=10.0
# Screen reading (`petctl read`) — Bolt OCRs a monitor and uses the text in
# his reply. Both optional: without them `petctl read` reports what's missing
# and the rest of the pet is unaffected.
# mss screen capture. X11/Win32/macOS — NOT Wayland.
# pytesseract a thin wrapper; the actual engine is a system package:
# sudo apt install tesseract-ocr
# No-sudo alternative to those two lines: pip install rapidocr-onnxruntime
# (pure pip, reuses the onnxruntime openwakeword already pulls in, slower to
# start). screen_text.resolve_engine() picks whichever is present.
mss>=9.0
pytesseract>=0.3.10
# Test runner (tests/ — pure logic, no audio hardware or display needed;
# run with QT_QPA_PLATFORM=offscreen).
pytest>=8.0
+779
View File
@@ -0,0 +1,779 @@
"""Draw Bolt — the pet — as per-state PNG frame sequences.
Produces the `assets/sprites/<state>/frame_NN.png` convention that
`bolt_pet/ui/sprite.py` loads (see `assets/sprites/README.md`). The art is
generated rather than sourced so it stays editable: tweak a colour or a pose
parameter here and re-run, instead of hand-editing 24 PNGs.
python scripts/generate_bolt_sprites.py # write into the real asset dir
python scripts/generate_bolt_sprites.py --out /tmp/prev # preview somewhere else
Everything is drawn in normalised 0..1 coordinates on a square canvas and
super-sampled `SS`x before being downscaled, because PIL's draw primitives
have no antialiasing of their own.
"""
from __future__ import annotations
import argparse
import math
from pathlib import Path
from PIL import Image, ImageDraw
SS = 4 # supersampling factor
OUT = 320 # final frame size (2x the default PET_SIZE of 160)
S = OUT * SS
# --- palette ---------------------------------------------------------------
# A cream shepherd-ish pup with a slate cap, amber eyes and a lightning blaze.
C_OUTLINE = (34, 42, 58, 255)
C_FUR = (246, 244, 238, 255)
C_FUR_SHADE = (214, 210, 200, 255)
C_DARK = (78, 92, 122, 255)
C_DARK2 = (58, 70, 96, 255)
C_INNER_EAR = (226, 154, 158, 255)
C_BROW = (206, 166, 118, 255)
# The far side of the walking pose. Distinctly darker than C_FUR_SHADE, which
# is too close to the cream to read as "behind the dog" at 160px.
C_FUR_FAR = (168, 176, 192, 255)
C_NOSE = (40, 48, 66, 255)
C_IRIS = (196, 128, 50, 255)
C_PUPIL = (30, 36, 50, 255)
C_WHITE = (255, 255, 255, 255)
C_BOLT = (255, 206, 61, 255)
C_COLLAR = (222, 84, 46, 255)
C_TAG = (255, 198, 68, 255)
C_TONGUE = (230, 116, 128, 255)
C_GLOW = (92, 214, 244, 255)
# --- layout constants (normalised) -----------------------------------------
HEAD_CX, HEAD_CY = 0.50, 0.375
HEAD_W, HEAD_H = 0.50, 0.44
NECK_Y = 0.565 # head layer rotates about here so tilts pivot at the neck
EAR_PIVOT = 0.335, 0.275
OW = 0.0105 # outline width, normalised
def px(v: float) -> float:
return v * S
def _w(width: float) -> int:
return max(1, int(round(px(width))))
def ell(d, cx, cy, w, h, fill, outline=C_OUTLINE, ow=OW):
d.ellipse(
[px(cx - w / 2), px(cy - h / 2), px(cx + w / 2), px(cy + h / 2)],
fill=fill,
outline=outline,
width=_w(ow) if outline else 0,
)
def rrect(d, cx, cy, w, h, r, fill, outline=C_OUTLINE, ow=OW):
d.rounded_rectangle(
[px(cx - w / 2), px(cy - h / 2), px(cx + w / 2), px(cy + h / 2)],
radius=px(r),
fill=fill,
outline=outline,
width=_w(ow) if outline else 0,
)
def poly(d, pts, fill, outline=C_OUTLINE, ow=OW):
d.polygon(
[(px(x), px(y)) for x, y in pts],
fill=fill,
outline=outline,
width=_w(ow) if outline else 0,
)
def rotate_pts(pts, pivot, deg):
a = math.radians(deg)
ca, sa = math.cos(a), math.sin(a)
ox, oy = pivot
out = []
for x, y in pts:
dx, dy = x - ox, y - oy
out.append((ox + dx * ca - dy * sa, oy + dx * sa + dy * ca))
return out
def lerp(a, b, t):
return a + (b - a) * t
def bolt_shape(cx, cy, w, h):
"""A lightning bolt polygon in a (w x h) box centred on (cx, cy)."""
unit = [
(0.62, 0.00),
(0.10, 0.56),
(0.44, 0.56),
(0.28, 1.00),
(0.90, 0.40),
(0.55, 0.40),
(0.80, 0.00),
]
return [(cx + (u - 0.5) * w, cy + (v - 0.5) * h) for u, v in unit]
# --- body ------------------------------------------------------------------
def _tail_points(p, steps=26):
"""Quadratic-bezier spine of the tail as (x, y, radius) samples.
Shared by the fill and outline passes so a wag can't move one and not the
other. The base sits deep inside the haunch, which is drawn over it, so
the tail reads as growing out of the body rather than floating beside it.
"""
wag = p["tail"]
base = (0.620, 0.845)
ctrl = (0.955, 0.870 - 0.025 * wag)
end = (0.905, 0.605 - 0.065 * wag)
pts = []
for i in range(steps + 1):
t = i / steps
x = (1 - t) ** 2 * base[0] + 2 * (1 - t) * t * ctrl[0] + t**2 * end[0]
y = (1 - t) ** 2 * base[1] + 2 * (1 - t) * t * ctrl[1] + t**2 * end[1]
pts.append((x, y, lerp(0.080, 0.042, t)))
return pts
def draw_tapered(d, pts, color_at):
"""Draw a tapered limb from (x, y, radius) samples.
Two passes: circles along the spine for the fill, then the two silhouette
edges, so it reads as one solid shape instead of a string of beads.
*color_at* takes 0..1 along the length, which is how the tail gets its
cream tip.
"""
last = len(pts) - 1
for i, (x, y, r) in enumerate(pts):
ell(d, x, y, r * 2, r * 2, color_at(i / last), outline=None)
for side in (1, -1):
edge = []
for i, (x, y, r) in enumerate(pts):
j = min(i + 1, last)
k = max(i - 1, 0)
tx, ty = pts[j][0] - pts[k][0], pts[j][1] - pts[k][1]
n = math.hypot(tx, ty) or 1e-6
nx, ny = -ty / n, tx / n
edge.append((x + nx * r * side, y + ny * r * side))
d.line([(px(x), px(y)) for x, y in edge], fill=C_OUTLINE, width=_w(OW), joint="curve")
x, y, r = pts[last]
ell(d, x, y, r * 2, r * 2, color_at(1.0))
def draw_tail(d, p):
# Only the last stretch is the cream tip. The haunch hides the first ~half
# of the tail, so a generous tip leaves the visible part looking like a
# pale blob floating next to the dog rather than its tail.
draw_tapered(d, _tail_points(p), lambda t: C_DARK if t < 0.84 else C_FUR)
def draw_body(d, p):
br = p["breathe"]
# haunches (sitting)
ell(d, 0.285, 0.795, 0.235, 0.275, C_DARK)
ell(d, 0.715, 0.795, 0.235, 0.275, C_DARK)
# torso
ell(d, 0.50, 0.745 - 0.004 * br, 0.455 + 0.012 * br, 0.395 + 0.014 * br, C_DARK)
# front legs
for cx in (0.415, 0.585):
rrect(d, cx, 0.845, 0.125, 0.215, 0.062, C_FUR)
ell(d, cx, 0.925, 0.155, 0.095, C_FUR)
# chest / belly blaze
ell(d, 0.50, 0.735 - 0.004 * br, 0.275 + 0.008 * br, 0.315 + 0.012 * br, C_FUR)
# toes
for cx in (0.415, 0.585):
for off in (-0.035, 0.0, 0.035):
d.arc(
[px(cx + off - 0.017), px(0.902), px(cx + off + 0.017), px(0.945)],
start=250,
end=290,
fill=C_FUR_SHADE,
width=_w(0.007),
)
def draw_collar(d, p):
rrect(d, 0.50, 0.585, 0.315, 0.062, 0.031, C_COLLAR)
tag = C_GLOW if p.get("tag_glow") else C_TAG
ell(d, 0.50, 0.638, 0.082, 0.082, tag)
poly(d, bolt_shape(0.50, 0.638, 0.030, 0.052), C_OUTLINE, outline=None)
# --- head ------------------------------------------------------------------
# Ear outline in a *local* frame: origin at the base on the skull, +x points
# outward (away from the muzzle), +y points up. Keeping it side-agnostic here
# and mirroring at draw time avoids sign confusion — an earlier version mixed
# the conventions and the ears flattened into a brim whenever they rotated.
_EAR_LOCAL = [
(-0.058, -0.038),
(0.078, -0.038),
(0.092, 0.140),
(0.030, 0.248),
(-0.038, 0.122),
]
def _ear_polygon(side, lean_deg):
"""Mirror + lean the local ear, returning canvas-space points.
*lean_deg* tips the ear away from vertical: 0 is fully perked, larger
values relax and eventually droop it out sideways.
"""
a = math.radians(lean_deg)
ca, sa = math.cos(a), math.sin(a)
pivot_x = 0.5 + side * (0.5 - EAR_PIVOT[0])
pts = []
for x, y in _EAR_LOCAL:
rx = x * ca + y * sa
ry = -x * sa + y * ca
pts.append((pivot_x + side * rx, EAR_PIVOT[1] - ry))
return pts
def draw_ears(d, p):
perk = p["ear"]
twitch = p.get("ear_twitch", 0.0)
for side in (-1, 1):
lean = 18.0 * (1.0 - perk) + 44.0 * max(0.0, -perk)
if side == 1:
lean -= twitch * 12.0
pts = _ear_polygon(side, lean)
poly(d, pts, C_DARK)
base_mid = (
(pts[0][0] + pts[1][0]) / 2,
(pts[0][1] + pts[1][1]) / 2,
)
inner = [(lerp(base_mid[0], x, 0.60), lerp(base_mid[1], y, 0.64)) for x, y in pts]
poly(d, inner, C_INNER_EAR, outline=None)
def draw_cap(layer, p):
"""Slate cap over the top of the head, clipped to the head silhouette."""
mask = Image.new("L", (S, S), 0)
ImageDraw.Draw(mask).ellipse(
[
px(HEAD_CX - HEAD_W / 2),
px(HEAD_CY - HEAD_H / 2),
px(HEAD_CX + HEAD_W / 2),
px(HEAD_CY + HEAD_H / 2),
],
fill=255,
)
cap = Image.new("RGBA", (S, S), (0, 0, 0, 0))
dc = ImageDraw.Draw(cap)
ell(dc, HEAD_CX, 0.245, 0.54, 0.30, C_DARK, outline=None)
# brow dip between the eyes, so the cap reads as a marking not a helmet
ell(dc, HEAD_CX, 0.352, 0.155, 0.115, C_FUR, outline=None)
cap.putalpha(Image.composite(cap.getchannel("A"), Image.new("L", (S, S), 0), mask))
layer.alpha_composite(cap)
def draw_eyes(d, p):
blink = p["blink"]
lx, ly = 0.383, 0.372
rx, ry = 0.617, 0.372
dx, dy = p.get("look", (0.0, 0.0))
for cx, cy in ((lx, ly), (rx, ry)):
if p.get("cross"):
for ang in (45, -45):
a = math.radians(ang)
hx, hy = 0.042 * math.cos(a), 0.042 * math.sin(a)
d.line(
[px(cx - hx), px(cy - hy), px(cx + hx), px(cy + hy)],
fill=C_OUTLINE,
width=_w(0.014),
)
continue
if blink > 0.55:
d.arc(
[px(cx - 0.052), px(cy - 0.030), px(cx + 0.052), px(cy + 0.040)],
start=200,
end=340,
fill=C_OUTLINE,
width=_w(0.013),
)
continue
h = lerp(0.118, 0.030, blink)
ell(d, cx, cy, 0.106, h, C_WHITE)
if h > 0.06:
ell(d, cx + dx, cy + dy * 0.6, 0.082, min(h - 0.022, 0.092), C_IRIS, outline=None)
ell(d, cx + dx, cy + dy * 0.6, 0.046, min(h - 0.045, 0.056), C_PUPIL, outline=None)
ell(d, cx + dx - 0.020, cy + dy * 0.6 - 0.024, 0.030, 0.026, C_WHITE, outline=None)
# Tan brow dots on the slate cap (the shepherd/doberman marking) rather
# than dashes — as lines above the eyes they read as heavy eyelids and
# make an idle pet look permanently fed up.
raise_ = p.get("brow", 0.0)
angry = p.get("brow_angle", 0.0)
for side, cx in ((-1, lx), (1, rx)):
by = 0.291 - 0.020 * raise_
ell(
d,
cx + side * 0.006,
by + side * angry * 0.020,
0.062,
0.040,
C_BROW,
outline=None,
)
def draw_muzzle(d, p):
mouth = p["mouth"]
ell(d, 0.50, 0.487, 0.285, 0.195, C_FUR)
# nose
ell(d, 0.50, 0.440, 0.105, 0.078, C_NOSE, outline=None)
ell(d, 0.478, 0.428, 0.030, 0.020, (92, 102, 124, 255), outline=None)
if mouth > 0.02:
h = 0.030 + 0.085 * mouth
w = 0.105 + 0.055 * mouth
ell(d, 0.50, 0.500 + h / 2 - 0.008, w, h, C_NOSE)
ell(d, 0.50, 0.500 + h * 0.72, w * 0.60, h * 0.52, C_TONGUE, outline=None)
else:
# closed muzzle: a short philtrum down from the nose into two
# downward-bulging curves (PIL arcs run clockwise from 3 o'clock with
# y down, so 0->180 is the lower half — the smiling side).
d.line([px(0.50), px(0.470), px(0.50), px(0.508)], fill=C_OUTLINE, width=_w(0.011))
for side in (-1, 1):
cx = 0.50 + side * 0.032
d.arc(
[px(cx - 0.032), px(0.492), px(cx + 0.032), px(0.536)],
start=0,
end=180,
fill=C_OUTLINE,
width=_w(0.011),
)
def draw_head(layer, p):
d = ImageDraw.Draw(layer)
draw_ears(d, p)
ell(d, HEAD_CX, HEAD_CY, HEAD_W, HEAD_H, C_FUR)
draw_cap(layer, p)
# blaze
poly(d, bolt_shape(0.50, 0.243, 0.088, 0.150), C_BOLT, outline=None)
draw_muzzle(d, p)
draw_eyes(d, p)
# --- extras ----------------------------------------------------------------
def draw_extras(layer, p):
d = ImageDraw.Draw(layer)
kind = p.get("extras")
if kind == "listen":
for i in range(3):
r = 0.045 + i * 0.036
alpha = int(210 - i * 55)
phase = p.get("phase", 0)
if (phase + i) % 3 == 0:
alpha = min(255, alpha + 45)
d.arc(
[px(0.845 - r), px(0.235 - r), px(0.845 + r), px(0.235 + r)],
start=200,
end=340,
fill=C_GLOW[:3] + (alpha,),
width=_w(0.014),
)
elif kind == "think":
phase = p.get("phase", 0)
for i in range(3):
grow = 1.0 if i == phase % 3 else 0.62
ell(
layer_d := d,
0.735 + i * 0.072,
0.145 - i * 0.030,
0.040 * grow,
0.040 * grow,
C_GLOW,
outline=C_OUTLINE,
ow=0.008,
)
elif kind == "error":
poly(d, bolt_shape(0.815, 0.185, 0.070, 0.120), (235, 92, 74, 255))
# --- side view: the walk cycle ---------------------------------------------
# The pose above is a front-facing sit, which is right for standing around but
# slides like a chess piece the moment the pet actually moves. Walking gets its
# own construction: a profile torso, four legs following a paw path, and a head
# side-on. Drawn facing RIGHT — ui/pet_window.py mirrors it when he walks left.
_GROUND = 0.930 # paw centre while a foot is planted
_STRIDE = 0.088 # how far ahead of / behind the pivot a paw reaches
_LIFT = 0.080 # peak height of a paw mid-swing
_STANCE = 0.62 # fraction of the cycle a foot spends on the ground
FRONT_PIVOT = (0.650, 0.620)
HIND_PIVOT = (0.315, 0.640)
WALK_HEAD = (0.780, 0.370, 0.260, 0.250) # cx, cy, w, h
def paw_position(pivot, phase):
"""Where one paw is at *phase* (0..1) of the cycle.
Stance is the half that matters: the foot is planted and travels backwards
under the dog at a constant rate. The window advances this cycle by
distance travelled rather than by clock, so that backwards travel cancels
the forward motion and the feet don't skate.
"""
phase %= 1.0
if phase < _STANCE:
t = phase / _STANCE
return pivot[0] + _STRIDE - 2 * _STRIDE * t, _GROUND
t = (phase - _STANCE) / (1.0 - _STANCE)
return (
pivot[0] - _STRIDE + 2 * _STRIDE * t,
_GROUND - _LIFT * math.sin(math.pi * t),
)
def draw_leg(d, pivot, paw, fill, bend=0.032, top=0.052, toe=0.030):
"""A limb from pivot to paw: a bezier through a displaced knee, tapered
from thigh to ankle.
Tapering matters more than it sounds a constant-width limb reads as a
length of white pipe, and four of them make the dog look like furniture.
"""
vx, vy = paw[0] - pivot[0], paw[1] - pivot[1]
length = math.hypot(vx, vy) or 1e-6
nx, ny = -vy / length, vx / length # perpendicular; points backwards
knee = (
(pivot[0] + paw[0]) / 2 + nx * bend,
(pivot[1] + paw[1]) / 2 + ny * bend,
)
pts = []
for i in range(13):
t = i / 12
x = (1 - t) ** 2 * pivot[0] + 2 * (1 - t) * t * knee[0] + t**2 * paw[0]
y = (1 - t) ** 2 * pivot[1] + 2 * (1 - t) * t * knee[1] + t**2 * paw[1]
pts.append((x, y, lerp(top, toe, t)))
draw_tapered(d, pts, lambda _t: fill)
ell(d, paw[0], paw[1] + 0.008, 0.098, 0.056, fill)
def draw_walk_tail(d, p, dy):
"""A curled plume over the back.
Cubic rather than quadratic: a single control point can only bend one way,
which gives a straight tapered tube a club with a white ball on the end,
not a tail. The curl back over the spine is what makes it read.
"""
wag = p["tail"]
base = (0.250, 0.575 + dy)
c1 = (0.075, 0.545 + dy - 0.030 * wag)
c2 = (0.070, 0.300 + dy - 0.040 * wag)
end = (0.215, 0.290 + dy - 0.020 * wag)
pts = []
for i in range(29):
t = i / 28
u = 1 - t
x = u**3 * base[0] + 3 * u**2 * t * c1[0] + 3 * u * t**2 * c2[0] + t**3 * end[0]
y = u**3 * base[1] + 3 * u**2 * t * c1[1] + 3 * u * t**2 * c2[1] + t**3 * end[1]
pts.append((x, y, lerp(0.076, 0.028, t)))
draw_tapered(d, pts, lambda t: C_DARK if t < 0.90 else C_FUR)
def draw_torso(d, dy):
"""Rump + barrel + chest as one silhouette.
Drawn in two passes every shape swollen by the stroke width in the
outline colour, then every shape again at true size in the fill. Outlining
each piece individually instead leaves the construction arcs showing
across the body, which looks like the dog has panel lines.
"""
shapes = [
("ell", 0.300, 0.600 + dy, 0.290, 0.300, 0.0),
("rrect", 0.480, 0.585 + dy, 0.520, 0.265, 0.130),
("ell", 0.650, 0.590 + dy, 0.250, 0.280, 0.0),
]
grow = 2 * OW
for colour, pad in ((C_OUTLINE, grow), (C_DARK, 0.0)):
for shape in shapes:
kind, cx, cy, w, h, extra = shape
if kind == "ell":
ell(d, cx, cy, w + pad, h + pad, colour, outline=None)
else:
rrect(d, cx, cy, w + pad, h + pad, extra + pad / 2, colour, outline=None)
# Belly kept small and low: any bigger and it merges with the cream legs
# into one white mass with a slate lid.
ell(d, 0.490, 0.672 + dy, 0.350, 0.098, C_FUR, outline=None)
def draw_walk_head(layer, p, dy):
d = ImageDraw.Draw(layer)
cx, cy, w, h = WALK_HEAD[0], WALK_HEAD[1] + dy, WALK_HEAD[2], WALK_HEAD[3]
# ear first, so the head covers its base
bounce = p.get("ear_bounce", 0.0)
ear = [
(0.690, cy - 0.030),
(0.700, cy - 0.150 - 0.012 * bounce),
(0.752, cy - 0.205 - 0.018 * bounce),
(0.788, cy - 0.090),
]
poly(d, ear, C_DARK)
inner = [(lerp(0.735, x, 0.58), lerp(cy - 0.040, y, 0.62)) for x, y in ear]
poly(d, inner, C_INNER_EAR, outline=None)
# neck into the chest
d.line(
[px(0.660), px(cy + 0.190), px(0.735), px(cy + 0.080)],
fill=C_OUTLINE,
width=_w(0.215),
joint="curve",
)
d.line(
[px(0.660), px(cy + 0.190), px(0.735), px(cy + 0.080)],
fill=C_DARK,
width=_w(0.190),
joint="curve",
)
ell(d, cx, cy, w, h, C_FUR)
# slate cap, clipped to the skull
mask = Image.new("L", (S, S), 0)
ImageDraw.Draw(mask).ellipse(
[px(cx - w / 2), px(cy - h / 2), px(cx + w / 2), px(cy + h / 2)], fill=255
)
cap = Image.new("RGBA", (S, S), (0, 0, 0, 0))
dc = ImageDraw.Draw(cap)
ell(dc, cx - 0.010, cy - 0.070, w * 1.02, h * 0.72, C_DARK, outline=None)
cap.putalpha(Image.composite(cap.getchannel("A"), Image.new("L", (S, S), 0), mask))
layer.alpha_composite(cap)
poly(d, bolt_shape(0.762, cy - 0.088, 0.062, 0.108), C_BOLT, outline=None)
# muzzle, nose, mouth
ell(d, 0.880, cy + 0.048, 0.145, 0.108, C_FUR)
ell(d, 0.950, cy + 0.018, 0.058, 0.048, C_NOSE, outline=None)
d.arc(
[px(0.885), px(cy + 0.058), px(0.945), px(cy + 0.100)],
start=0,
end=150,
fill=C_OUTLINE,
width=_w(0.010),
)
# one eye in profile, plus the brow marking
ell(d, 0.812, cy - 0.020, 0.092, 0.100, C_WHITE)
ell(d, 0.820, cy - 0.020, 0.062, 0.070, C_IRIS, outline=None)
ell(d, 0.824, cy - 0.020, 0.036, 0.042, C_PUPIL, outline=None)
ell(d, 0.812, cy - 0.040, 0.026, 0.022, C_WHITE, outline=None)
ell(d, 0.795, cy - 0.088, 0.055, 0.034, C_BROW, outline=None)
# Collar: a band *across* the neck, so it has to run perpendicular to it.
# Along the neck it just reads as an orange brick stuck to his chest.
collar = [
(px(0.648), px(cy + 0.098)),
(px(0.762), px(cy + 0.196)),
]
d.line(collar, fill=C_OUTLINE, width=_w(0.070), joint="curve")
d.line(collar, fill=C_COLLAR, width=_w(0.050), joint="curve")
ell(d, 0.712, cy + 0.196, 0.070, 0.070, C_TAG)
poly(d, bolt_shape(0.712, cy + 0.196, 0.025, 0.044), C_OUTLINE, outline=None)
def render_walk_frame(p) -> Image.Image:
base = Image.new("RGBA", (S, S), (0, 0, 0, 0))
d = ImageDraw.Draw(base)
phase = p["phase"]
# Two contacts per cycle, so the body dips twice — the give-away that a
# walk cycle is weight-bearing rather than a slide.
dy = -0.011 * abs(math.sin(2 * math.pi * phase))
head_dy = dy * 0.6 - 0.006 * math.sin(2 * math.pi * phase + 0.7)
# Diagonal pairs (a trot): each front leg moves with the opposite hind.
far_front = paw_position(FRONT_PIVOT, phase + 0.5)
far_hind = paw_position(HIND_PIVOT, phase)
near_front = paw_position(FRONT_PIVOT, phase)
near_hind = paw_position(HIND_PIVOT, phase + 0.5)
draw_walk_tail(d, p, dy)
# far side first, in the shade colour, so the near legs read as in front
draw_leg(d, (HIND_PIVOT[0], HIND_PIVOT[1] + dy), far_hind, C_FUR_FAR, bend=0.046)
draw_leg(d, (FRONT_PIVOT[0], FRONT_PIVOT[1] + dy), far_front, C_FUR_FAR)
draw_torso(d, dy)
draw_leg(d, (HIND_PIVOT[0], HIND_PIVOT[1] + dy), near_hind, C_FUR, bend=0.046)
draw_leg(d, (FRONT_PIVOT[0], FRONT_PIVOT[1] + dy), near_front, C_FUR)
draw_walk_head(base, p, head_dy)
return base.resize((OUT, OUT), Image.LANCZOS)
# --- frame assembly --------------------------------------------------------
def default_pose(**over):
p = dict(
breathe=0.0,
tail=0.0,
ear=0.0,
ear_twitch=0.0,
blink=0.0,
mouth=0.0,
tilt=0.0,
head_dy=0.0,
look=(0.0, 0.0),
brow=0.0,
brow_angle=0.0,
cross=False,
tag_glow=False,
extras=None,
phase=0,
)
p.update(over)
return p
def render_frame(p) -> Image.Image:
if p.get("pose") == "walk":
return render_walk_frame(p)
base = Image.new("RGBA", (S, S), (0, 0, 0, 0))
body = Image.new("RGBA", (S, S), (0, 0, 0, 0))
db = ImageDraw.Draw(body)
draw_tail(db, p)
draw_body(db, p)
draw_collar(db, p)
base.alpha_composite(body)
head = Image.new("RGBA", (S, S), (0, 0, 0, 0))
draw_head(head, p)
if p["tilt"]:
head = head.rotate(
p["tilt"], resample=Image.BICUBIC, center=(px(HEAD_CX), px(NECK_Y))
)
dy = int(px(p["head_dy"]))
if dy:
shifted = Image.new("RGBA", (S, S), (0, 0, 0, 0))
shifted.alpha_composite(head, (0, dy))
head = shifted
base.alpha_composite(head)
draw_extras(base, p)
return base.resize((OUT, OUT), Image.LANCZOS)
def frames_for(state: str) -> list[dict]:
if state == "idle":
out = []
for i in range(8):
t = i / 8
br = math.sin(t * 2 * math.pi)
out.append(
default_pose(
breathe=br,
head_dy=-0.006 * br,
tail=math.sin(t * 4 * math.pi),
blink=1.0 if i == 6 else 0.0,
)
)
return out
if state == "listening":
out = []
for i in range(4):
t = i / 4
out.append(
default_pose(
ear=1.0,
ear_twitch=0.35 * math.sin(t * 2 * math.pi),
tilt=-7 + 2.0 * math.sin(t * 2 * math.pi),
brow=1.0,
tail=0.5 * math.sin(t * 2 * math.pi),
head_dy=-0.008,
tag_glow=True,
extras="listen",
phase=i,
)
)
return out
if state == "thinking":
out = []
for i in range(6):
t = i / 6
out.append(
default_pose(
ear=0.25,
tilt=6.0,
look=(0.022, -0.026),
brow=0.5,
breathe=0.4 * math.sin(t * 2 * math.pi),
tail=0.2 * math.sin(t * 2 * math.pi),
extras="think",
phase=i // 2,
)
)
return out
if state == "talking":
out = []
for i in range(4):
t = i / 4
open_ = (math.sin(t * 2 * math.pi) + 1) / 2
out.append(
default_pose(
mouth=0.25 + 0.75 * open_,
ear=0.6,
head_dy=-0.010 * open_,
breathe=open_,
tail=math.sin(t * 2 * math.pi + 1.0),
brow=0.35,
)
)
return out
if state == "walk":
# 8 frames: two full strides, so the loop lands back on the pose it
# started from and the cycle is seamless however it's entered.
out = []
for i in range(8):
phase = i / 8
out.append(
default_pose(
pose="walk",
phase=phase,
tail=math.sin(2 * math.pi * phase),
ear_bounce=math.sin(2 * math.pi * phase + 0.9),
)
)
return out
if state == "error":
return [
default_pose(ear=-1.0, cross=True, brow_angle=1.0, mouth=0.35, tail=-0.6,
extras="error"),
default_pose(ear=-0.85, cross=True, brow_angle=1.0, mouth=0.15, tail=-0.4,
head_dy=0.008),
]
raise ValueError(state)
STATES = ["idle", "listening", "thinking", "talking", "error", "walk"]
def main():
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument(
"--out",
type=Path,
default=Path(__file__).resolve().parent.parent / "bolt_pet" / "assets" / "sprites",
)
ap.add_argument("--states", nargs="*", default=STATES)
args = ap.parse_args()
for state in args.states:
d = args.out / state
d.mkdir(parents=True, exist_ok=True)
for old in d.glob("*.png"):
old.unlink()
for i, pose in enumerate(frames_for(state)):
render_frame(pose).save(d / f"frame_{i:02d}.png")
print(f"{state}: {len(frames_for(state))} frames -> {d}")
if __name__ == "__main__":
main()
+155 -2
View File
@@ -1,4 +1,5 @@
"""Barge-in detection, driven by a fake mic stream (no audio hardware)."""
"""Barge-in detection, driven by a fake mic stream and a fake wake model
(no audio hardware, no ONNX runtime)."""
import sys
from pathlib import Path
@@ -7,7 +8,7 @@ import numpy as np
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.audio.barge_in import BargeInDetector
from bolt_pet.audio.barge_in import BargeInDetector, WakeWordBargeIn, make_detector
class FakeStream:
@@ -63,3 +64,155 @@ def test_a_mic_error_mid_playback_is_not_fatal():
detector = BargeInDetector(BrokenStream(), threshold=1000, required_frames=1)
assert detector.check() is False
# ── wake-word mode ───────────────────────────────────────────────────────────
class FakeModel:
"""Scores frames from a canned list, mimicking openWakeWord's
{class_name: score} return. Records reset() calls."""
def __init__(self, scores):
self._scores = list(scores)
self.resets = 0
def predict(self, frame):
score = self._scores.pop(0) if self._scores else 0.0
return {"thunderbolt": score}
def reset(self):
self.resets += 1
def _wake_detector(scores, threshold=0.5, amplitudes=None):
stream = FakeStream(amplitudes if amplitudes is not None else [500] * len(scores))
return WakeWordBargeIn(stream, model=FakeModel(scores), threshold=threshold), stream
def test_loud_noise_alone_does_not_interrupt_in_wake_mode():
"""The whole point of wake mode: a slammed door is deafening and scores
nothing, so the pet keeps talking."""
detector, _ = _wake_detector([0.01] * 6, amplitudes=[30000] * 6)
assert not any(detector.check() for _ in range(6))
def test_the_wake_word_interrupts():
detector, _ = _wake_detector([0.1, 0.2, 0.9])
assert [detector.check() for _ in range(3)] == [False, False, True]
def test_a_single_frame_is_enough_when_it_clears_the_threshold():
detector, _ = _wake_detector([0.55])
assert detector.check() is True
def test_scores_just_under_the_threshold_do_not_fire():
detector, _ = _wake_detector([0.49, 0.499], threshold=0.5)
assert not any(detector.check() for _ in range(2))
def test_detecting_resets_the_model_so_the_tail_is_not_reused():
model = FakeModel([0.9])
detector = WakeWordBargeIn(FakeStream([500]), model=model, threshold=0.5)
assert detector.check() is True
assert model.resets == 1
def test_reset_clears_the_models_audio_window_not_just_predictions():
"""The regression that made the pet interrupt itself a word into every
reply: openwakeword's reset() clears only the prediction buffer, so the
"thunderbolt" that started the turn was still in the preprocessor's
rolling window when playback began, and the first frame fed to the model
re-fired on it."""
class FakePreprocessor:
def __init__(self):
self.raw_data_buffer = [1, 2, 3]
self.feature_buffer = np.ones((120, 96))
self.melspectrogram_buffer = np.zeros((76, 32))
self.accumulated_samples = 4096
def _get_embeddings(self, audio):
return np.zeros((120, 96))
model = FakeModel([0.9])
model.preprocessor = FakePreprocessor()
WakeWordBargeIn(FakeStream([500]), model=model, threshold=0.5).reset()
assert model.preprocessor.raw_data_buffer == []
assert model.preprocessor.accumulated_samples == 0
assert not model.preprocessor.feature_buffer.any() # blank, not the old audio
assert model.preprocessor.melspectrogram_buffer.all() # restored to ones
def test_the_lazy_wrapper_exposes_its_preprocessor():
"""The pet holds _default_model — a lazy *wrapper* around openwakeword's
Model. If the wrapper stops proxying .preprocessor, hard_reset() finds
nothing to clear and silently degrades to the shallow reset that leaves
the last detection in the audio window. That failure is invisible: no
exception, no log, the pet just interrupts itself again."""
from bolt_pet.audio.wake_word import _OpenWakeWordModel
wrapper = _OpenWakeWordModel()
assert hasattr(wrapper, "preprocessor")
assert wrapper.preprocessor is None # not loaded yet: a no-op, not a load
class FakeInner:
preprocessor = object()
def reset(self):
pass
wrapper._model = FakeInner()
assert wrapper.preprocessor is FakeInner.preprocessor
def test_a_model_without_a_preprocessor_still_resets():
"""Fakes in tests, and any future openwakeword whose internals moved."""
model = FakeModel([0.0])
WakeWordBargeIn(FakeStream([500]), model=model, threshold=0.5).reset()
assert model.resets == 1
def test_a_callable_threshold_is_read_every_frame():
"""The tray tuner's slider has to apply mid-playback, not just mid-idle."""
threshold = {"value": 0.9}
detector = WakeWordBargeIn(
FakeStream([500] * 2), model=FakeModel([0.6, 0.6]),
threshold=lambda: threshold["value"],
)
assert detector.check() is False
threshold["value"] = 0.5
assert detector.check() is True
def test_a_model_that_blows_up_mid_playback_is_not_fatal():
class BrokenModel:
def predict(self, frame):
raise RuntimeError("onnx session died")
def reset(self):
raise RuntimeError("still dead")
detector = WakeWordBargeIn(FakeStream([500]), model=BrokenModel(), threshold=0.5)
assert detector.check() is False
detector.reset() # must not raise either
def test_a_mic_error_is_not_fatal_in_wake_mode():
class BrokenStream:
def read(self, frames):
raise OSError("device disappeared")
detector = WakeWordBargeIn(BrokenStream(), model=FakeModel([0.9]), threshold=0.5)
assert detector.check() is False
def test_make_detector_picks_the_mode():
stream = FakeStream([0])
assert isinstance(make_detector(stream, mode="wake"), WakeWordBargeIn)
assert isinstance(make_detector(stream, mode="energy"), BargeInDetector)
# A typo in .env shouldn't stop the pet from starting.
assert isinstance(make_detector(stream, mode="waek"), BargeInDetector)
+3 -3
View File
@@ -41,10 +41,10 @@ def test_full_turn_happy_path(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's the weather")
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: "sunny and 72")
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: controller_mod.server_client.Reply("sunny and 72"))
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
ctrl._handle_conversation_turn()
@@ -150,7 +150,7 @@ def test_heartbeat_speaks_a_pending_announcement_when_idle(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: "don't forget your 3pm")
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
said = _capture(ctrl.said)
ctrl._maybe_heartbeat()
+771 -10
View File
@@ -5,6 +5,7 @@ the live wake threshold.
Needs a QApplication (signals), so run with QT_QPA_PLATFORM=offscreen.
"""
import json
import sys
from pathlib import Path
@@ -14,6 +15,7 @@ from PySide6.QtWidgets import QApplication
from bolt_pet import controller as controller_mod
from bolt_pet.notifications import Notification
from bolt_pet.server_client import Reply
from bolt_pet.state import PetState
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
@@ -79,12 +81,106 @@ def test_petctl_nap_also_flips_the_controller_state(ctrl):
assert ctrl._napping is True
# ── filectl routing ──────────────────────────────────────────────────────────
# filectl's wire format is a single-line JSON envelope (see file_ops.py's
# module docstring for why) — build commands with json.dumps so the tests
# don't hardcode escaping by hand.
def _filectl(payload: dict) -> str:
return "filectl " + json.dumps(payload)
def test_filectl_commands_never_reach_the_shell(monkeypatch, ctrl, tmp_path):
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
target = tmp_path / "a.txt"
output = ctrl._handle_command(_filectl({"op": "write", "path": str(target), "content": "hello"}))
assert ran == []
assert target.read_text() == "hello"
assert str(target) in output
def test_filectl_edit_round_trips_through_the_relay(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.server_client, "run_local_command",
lambda cmd: (_ for _ in ()).throw(AssertionError("should not shell out")))
target = tmp_path / "a.py"
target.write_text("x = 1\n")
output = ctrl._handle_command(
_filectl({"op": "edit", "path": str(target), "old": "x = 1", "new": "x = 2"})
)
assert target.read_text() == "x = 2\n"
assert str(target) in output
def test_bad_filectl_syntax_is_reported_back_not_executed(monkeypatch, ctrl):
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
output = ctrl._handle_command("filectl {not valid json")
assert ran == []
assert "[filectl]" in output
def test_filectl_execution_failure_is_reported_back_not_raised(monkeypatch, ctrl):
output = ctrl._handle_command(_filectl({"op": "read", "path": "/no/such/file.txt"}))
assert "[filectl]" in output
def test_ordinary_commands_still_run_locally_alongside_filectl(monkeypatch, ctrl):
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command",
lambda cmd: ran.append(cmd) or "[exit 0]\n")
ctrl._handle_command("df -h /")
assert ran == ["df -h /"]
# ── barge-in ────────────────────────────────────────────────────────────────
def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch, ctrl):
"""_speak resets the detector after playback so the pet's own voice
doesn't linger in the wake model's window. Reading the stats after that
reset reports 0.000 at frame 0 for every interruption, which is worse
than no instrumentation it looks like hard evidence and isn't."""
from bolt_pet.audio import barge_in as barge_in_mod
class Detector(barge_in_mod.WakeWordBargeIn):
def __init__(self):
self._frames, self._peak, self._last, self._last_threshold = 0, 0.0, 0.0, 0.5
self._frame_len = 1280
self.reset_calls = 0
def reset(self):
self.reset_calls += 1
self._frames, self._peak, self._last = 0, 0.0, 0.0
detector = Detector()
ctrl._barge_in = detector
logs = _capture(ctrl.log)
def interrupted_playback(text, on_error=None, should_stop=None, voice_id=None):
# What really happens: frames get scored during playback, then one
# clears the threshold and playback aborts.
detector._frames, detector._peak, detector._last = 7, 0.81, 0.81
return False
monkeypatch.setattr(controller_mod.tts, "speak", interrupted_playback)
ctrl._speak("a very long explanation")
interrupted = next(m for m in logs if "Interrupted" in m)
assert "0.810" in interrupted and "frame 7" in interrupted
# Once before playback (clear the window) and once after (drop the pet's
# own voice) — the point is that the *read* happens between them.
assert detector.reset_calls == 2
def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
logs = _capture(ctrl.log)
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: False) # interrupted
lambda text, on_error=None, should_stop=None, voice_id=None: False) # interrupted
ctrl._speak("a very long explanation")
@@ -94,14 +190,121 @@ def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
def test_uninterrupted_playback_does_not_queue_a_turn(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
ctrl._speak("short answer")
assert not ctrl._talk_now.is_set()
# ── follow-up listening ─────────────────────────────────────────────────────
@pytest.fixture
def spoke(monkeypatch):
"""Playback that always completes, so only the follow-up rule decides
whether another turn is queued."""
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None: True)
def test_a_reply_ending_in_a_question_keeps_listening(spoke, ctrl):
logs = _capture(ctrl.log)
ctrl._speak("You're still in ~/Documents/bolt-pet. Ready to run a command?")
assert ctrl._talk_now.is_set() # no wake word needed for the answer
assert ctrl._pending_follow_up # and the next turn knows it's an answer
assert any("listening for your answer" in message for message in logs)
def test_a_statement_does_not_keep_listening(spoke, ctrl):
ctrl._speak("It's 7:15 AM on July 23, 2026.")
assert not ctrl._talk_now.is_set()
assert not ctrl._pending_follow_up
def test_a_question_anywhere_in_the_reply_keeps_listening(spoke, ctrl):
"""Bolt often asks and then keeps talking ("Want me to fix it? I'd start
with the config."), so the question mark doesn't have to be last. An
unwanted extra listen ends itself on VAD_GRACE_SECONDS of silence; a missed
one costs you a wake word, which is the more expensive mistake."""
ctrl._speak("Want me to restart it? It's been up for 40 days.")
assert ctrl._talk_now.is_set()
def test_follow_ups_stop_at_the_cap(spoke, monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "FOLLOW_UP_MAX_TURNS", 2)
logs = _capture(ctrl.log)
for _ in range(2):
ctrl._speak("Want me to keep going?")
ctrl._talk_now.clear()
assert ctrl._follow_ups == 2
ctrl._speak("Want me to keep going?")
assert not ctrl._talk_now.is_set() # chain broken until you re-trigger it
assert any("Follow-up limit reached" in message for message in logs)
def test_the_cap_is_not_announced_on_ordinary_replies(spoke, monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "FOLLOW_UP_MAX_TURNS", 1)
ctrl._follow_ups = 1
logs = _capture(ctrl.log)
ctrl._speak("Done — the file is saved.")
assert not any("Follow-up limit" in message for message in logs)
def test_starting_a_turn_yourself_resets_the_chain(spoke, monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **kw: None) # you said nothing
ctrl._follow_ups = 3
ctrl._handle_conversation_turn()
assert ctrl._follow_ups == 0
def test_an_answered_question_gets_a_longer_grace_period(spoke, monkeypatch, ctrl):
"""You were just asked something — you get longer to think than when you
deliberately said the wake word."""
grace = []
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **kw: grace.append(kw.get("grace_s")) or None)
ctrl._handle_conversation_turn() # you started this one
ctrl._pending_follow_up = True
ctrl._handle_conversation_turn() # this one answers a question
assert grace == [None, controller_mod.config.FOLLOW_UP_GRACE_SECONDS]
def test_muting_stops_follow_ups(spoke, ctrl):
ctrl._muted = True
ctrl._speak("Shall I continue?")
assert not ctrl._talk_now.is_set() # mute means don't listen, question or not
def test_follow_up_can_be_turned_off(spoke, monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "FOLLOW_UP_LISTEN", False)
ctrl._speak("Shall I continue?")
assert not ctrl._talk_now.is_set()
def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None: False)
ctrl._follow_ups = 3
ctrl._speak("a very long explanation")
assert ctrl._follow_ups == 0 # you're clearly engaged
assert ctrl._talk_now.is_set()
def test_speech_is_recorded_in_the_history(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
ctrl._speak("**bold** reply")
assert ctrl.history.last().text == "**bold** reply" # raw, for copy/paste
@@ -115,10 +318,10 @@ def test_the_active_window_rides_along_with_the_utterance(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.screen_context, "context_for",
lambda text: f"{text}\n\n[on screen right now: app.py]")
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or "that's a KeyError")
lambda text, on_command=None: sent.append(text) or Reply("that's a KeyError"))
ctrl._handle_conversation_turn()
@@ -171,10 +374,10 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "you awake?")
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: "always")
lambda text, on_command=None: Reply("always"))
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
ctrl.set_napping(True)
ctrl._handle_conversation_turn()
@@ -182,6 +385,148 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
assert spoken == ["always"]
# ── local intents ───────────────────────────────────────────────────────────
@pytest.fixture
def heard(monkeypatch):
"""A turn where you said something, with the server and TTS recorded.
Returns (utterance_setter, sent, spoken)."""
said = {"text": ""}
sent, spoken = [], []
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said["text"])
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or Reply("from the server"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None:
spoken.append(text) or True)
return said, sent, spoken
def test_a_local_intent_never_reaches_the_server(heard, ctrl):
said, sent, spoken = heard
said["text"] = "come here"
actions = _capture(ctrl.action)
ctrl._handle_conversation_turn()
assert sent == [] # no round trip at all
assert actions == [{"action": "move", "anchor": "cursor"}]
assert spoken == [] # walking over is the reply
assert ctrl._state.state == PetState.IDLE
def test_stop_is_answered_with_silence(heard, ctrl):
said, sent, spoken = heard
said["text"] = "be quiet"
ctrl._handle_conversation_turn()
assert (sent, spoken) == ([], [])
def test_a_request_that_merely_starts_with_an_intent_word_goes_to_the_server(heard, ctrl):
said, sent, spoken = heard
said["text"] = "stop the docker container"
ctrl._handle_conversation_turn()
assert sent and "stop the docker container" in sent[0]
assert spoken == ["from the server"]
def test_local_intents_are_skipped_while_answering_a_question(heard, ctrl):
"""Bolt asked something; "never mind" is an answer to him, not a body
command. Swallowing it locally would leave the server holding a question it
never got a reply to."""
said, sent, spoken = heard
said["text"] = "never mind"
ctrl._pending_follow_up = True
ctrl._handle_conversation_turn()
assert sent and "never mind" in sent[0]
def test_local_intents_can_be_turned_off(monkeypatch, heard, ctrl):
monkeypatch.setattr(controller_mod.config, "LOCAL_INTENTS", False)
said, sent, spoken = heard
said["text"] = "come here"
ctrl._handle_conversation_turn()
assert sent and "come here" in sent[0]
def test_say_that_again_replays_the_last_line_without_duplicating_history(heard, ctrl):
said, sent, spoken = heard
ctrl.history.add(controller_mod.history_mod.PET, "it's 7:15 AM", 0.0)
said["text"] = "what did you say?"
ctrl._handle_conversation_turn()
assert spoken == ["it's 7:15 AM"]
assert sent == []
pet_lines = [e.text for e in ctrl.history.entries()
if e.role == controller_mod.history_mod.PET]
assert pet_lines == ["it's 7:15 AM"] # replayed, not re-recorded
def test_repeat_with_nothing_to_repeat_says_so(heard, ctrl):
said, sent, spoken = heard
said["text"] = "say that again"
ctrl._handle_conversation_turn()
assert spoken == ["I haven't said anything yet."]
def test_going_back_to_the_normal_voice_needs_no_server_prompt_support(heard, ctrl):
"""The server can only offer `petctl voice reset` if its prompt happens to
advertise the verb; recognising the phrase here works regardless."""
said, sent, spoken = heard
ctrl._voice_id, ctrl._voice_name = "voice-123", "Brian"
changed = _capture(ctrl.voice_changed)
said["text"] = "go back to your normal voice"
ctrl._handle_conversation_turn()
assert ctrl._voice_id == ""
assert changed == [""]
assert spoken == ["Back to my own voice."]
assert sent == []
def test_go_to_sleep_overrides_the_quiet_hours_schedule(heard, ctrl):
said, sent, spoken = heard
said["text"] = "go to sleep"
ctrl._handle_conversation_turn()
assert ctrl._napping is True
assert ctrl._nap_forced is True # not undone by the next schedule check
assert spoken == ["Night."]
# ── failure containment ─────────────────────────────────────────────────────
def test_one_bad_turn_does_not_end_the_session(monkeypatch, ctrl):
"""A turn raising something unforeseen used to unwind _loop and kill the
thread the pet would go deaf until it was restarted by hand."""
logs = _capture(ctrl.log)
def explode():
raise RuntimeError("numpy said no")
assert ctrl._guarded(explode, "conversation turn") is False
assert ctrl._state.state == PetState.IDLE
assert any("Recovered from a conversation turn failure" in m for m in logs)
def test_a_command_handler_crash_is_reported_up_the_relay(monkeypatch, ctrl):
"""The server is blocked on /desk/tool_result while this runs. Raising would
leave it waiting out its own timeout on a turn that can never finish."""
monkeypatch.setattr(controller_mod.pet_actions, "parse",
lambda command: (_ for _ in ()).throw(KeyError("boom")))
output = ctrl._handle_command("petctl move top-left")
assert output.startswith("[error]") and "boom" in output
# ── notification bridge ─────────────────────────────────────────────────────
def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
@@ -189,9 +534,9 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
sent, spoken = [], []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or "your build is green")
lambda text, on_command=None: sent.append(text) or Reply("your build is green"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
ctrl._drain_notifications()
@@ -204,7 +549,54 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
def test_filtered_out_notifications_are_never_queued(ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("deploy", 0)
ctrl._queue_notification(Notification(app="Chat", summary="lunch?", body=""))
assert ctrl._pending_notifications == []
assert not ctrl._pending_notifications
def test_the_notification_queue_is_bounded(monkeypatch, ctrl):
"""An overnight nap can't grow the queue without limit — the drain only runs
from the heartbeat, and the heartbeat doesn't run while napping."""
monkeypatch.setattr(controller_mod.config, "NOTIFICATION_QUEUE_LIMIT", 3)
ctrl = controller_mod.PetController()
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
for index in range(10):
ctrl._queue_notification(Notification(app="CI", summary=f"build {index}", body=""))
queued = [notification.summary for _stamp, notification in ctrl._pending_notifications]
assert queued == ["build 7", "build 8", "build 9"] # oldest dropped
def test_stale_notifications_are_dropped_instead_of_read_out(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "NOTIFICATION_MAX_AGE_SECONDS", 900)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or Reply(""))
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
ctrl._queue_notification(Notification(app="CI", summary="fresh", body=""))
# Backdate it past the age limit, as an overnight backlog would be.
stamp, notification = ctrl._pending_notifications.pop()
ctrl._pending_notifications.append((stamp - 4000, notification))
ctrl._drain_notifications()
assert sent == []
def test_a_nap_starting_mid_drain_keeps_the_rest_queued(monkeypatch, ctrl):
"""The old code swapped the queue out and returned, losing the remainder."""
def converse(text, on_command=None):
ctrl._napping = True # e.g. quiet hours began, or a fullscreen app opened
return Reply("")
monkeypatch.setattr(controller_mod.server_client, "converse", converse)
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
for index in range(3):
ctrl._queue_notification(Notification(app="CI", summary=f"build {index}", body=""))
ctrl._drain_notifications()
remaining = [notification.summary for _stamp, notification in ctrl._pending_notifications]
assert remaining == ["build 1", "build 2"]
def test_notifications_are_not_forwarded_while_napping(monkeypatch, ctrl):
@@ -241,3 +633,372 @@ def test_near_misses_are_recorded_for_the_tuner(ctrl):
ctrl.reset_wake_stats()
assert ctrl.wake_stats()["near_misses"] == []
# ── file delivery ────────────────────────────────────────────────────────────
def test_check_deliveries_downloads_and_saves_queued_files(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files",
lambda: [{"id": "abc", "name": "report.pdf", "size": 5}])
monkeypatch.setattr(controller_mod.server_client, "download_outbox_file",
lambda file_id: b"hello" if file_id == "abc" else b"")
ctrl._check_deliveries()
assert (tmp_path / "report.pdf").read_bytes() == b"hello"
def test_check_deliveries_is_a_noop_when_nothing_queued(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files", lambda: [])
downloaded = []
monkeypatch.setattr(controller_mod.server_client, "download_outbox_file",
lambda file_id: downloaded.append(file_id))
ctrl._check_deliveries()
assert downloaded == []
def test_check_deliveries_can_be_disabled(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "RECEIVE_FILES", False)
called = {"n": 0}
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files",
lambda: called.__setitem__("n", called["n"] + 1))
ctrl._check_deliveries()
assert called["n"] == 0
def test_check_deliveries_logs_and_continues_on_download_failure(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files", lambda: [
{"id": "bad", "name": "a.txt", "size": 1},
{"id": "good", "name": "b.txt", "size": 1},
])
def fake_download(file_id):
if file_id == "bad":
raise controller_mod.server_client.ServerError("gone")
return b"ok"
monkeypatch.setattr(controller_mod.server_client, "download_outbox_file", fake_download)
logs = _capture(ctrl.log)
ctrl._check_deliveries()
assert (tmp_path / "b.txt").read_bytes() == b"ok"
assert not (tmp_path / "a.txt").exists()
assert any("bad" in msg or "a.txt" in msg for msg in logs)
def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "send me that file")
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: Reply("it's on the way"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None: True)
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files",
lambda: [{"id": "abc", "name": "notes.txt", "size": 2}])
monkeypatch.setattr(controller_mod.server_client, "download_outbox_file",
lambda file_id: b"hi")
ctrl._handle_conversation_turn()
assert (tmp_path / "notes.txt").read_bytes() == b"hi"
# ── server-picked voice (the desk API's speak_as marker) ────────────────────
def _voice_turn(monkeypatch, ctrl, reply, said="talk like a pirate"):
"""Run one full conversation turn whose reply is *reply*, returning the
voice_id each tts.speak() call was given."""
voices = []
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said)
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: reply)
monkeypatch.setattr(
controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None:
voices.append(voice_id) or True,
)
ctrl._handle_conversation_turn()
return voices
def test_a_speak_as_reply_is_spoken_in_that_voice(monkeypatch, ctrl):
voices = _voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
assert voices == ["VOICE1"]
assert ctrl.current_voice() == "Terence"
def test_the_picked_voice_sticks_for_later_replies(monkeypatch, ctrl):
"""The server tags one reply and doesn't keep the id in its history, so
it can't re-request the voice when you say "keep talking like that"."""
monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", True)
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
voices = _voice_turn(monkeypatch, ctrl, Reply("Still me."), said="and now?")
assert voices == ["VOICE1"]
def test_voice_stickiness_can_be_turned_off(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "VOICE_STICKY", False)
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
voices = _voice_turn(monkeypatch, ctrl, Reply("Back to normal."), said="and now?")
assert voices == [None]
assert ctrl.current_voice() == ""
def test_a_new_pick_replaces_the_old_one(monkeypatch, ctrl):
changes = _capture(ctrl.voice_changed)
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
voices = _voice_turn(monkeypatch, ctrl, Reply("こんにちは。", "VOICE2", "Asahi"),
said="say that in Japanese")
assert voices == ["VOICE2"]
assert changes == ["Terence", "Asahi"]
def test_resetting_the_voice_goes_back_to_the_default(monkeypatch, ctrl):
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
changes = _capture(ctrl.voice_changed)
ctrl.reset_voice()
assert ctrl.current_voice() == ""
assert changes == [""] # the tray's menu entry follows this signal
voices = _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi")
assert voices == [None]
def test_an_unnamed_voice_still_reports_something_resettable(monkeypatch, ctrl):
"""voice_name is optional server-side — falling back to the id keeps the
tray entry from reading "now: " with nothing after it."""
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1"))
assert ctrl.current_voice() == "VOICE1"
def test_petctl_voice_reset_returns_bolt_to_his_own_voice(monkeypatch, ctrl):
"""The server can pick a voice but can't ask for the default back — it
was never told what Bolt's own voice id is. This is how it asks."""
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
_voice_turn(monkeypatch, ctrl, Reply("Ahoy.", "VOICE1", "Terence"))
output = ctrl._handle_command("petctl voice reset")
assert ran == [] # never reaches a shell, like every other petctl verb
assert "Terence" in output # the server can't see the voice; tell it what changed
assert ctrl.current_voice() == ""
assert _voice_turn(monkeypatch, ctrl, Reply("Normal again."), said="hi") == [None]
def test_petctl_voice_reset_says_so_when_there_was_nothing_to_reset(ctrl):
assert "already" in ctrl._handle_command("petctl voice reset")
# ── dialoguectl (multi-voice scenes) ────────────────────────────────────────
def _dialogue_command(*lines):
import json
return "dialoguectl " + json.dumps({"lines": list(lines)})
def test_dialoguectl_never_reaches_the_shell(monkeypatch, ctrl):
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm",
lambda pcm, rate, should_stop=None: True)
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "[cheerfully] hi"}))
assert ran == []
assert "[dialogue] played 1 line" in output
def test_a_scene_shows_in_the_bubble_with_the_delivery_tags_stripped(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
said = _capture(ctrl.said)
ctrl._handle_command(_dialogue_command(
{"voice": "self", "text": "[cheerfully] Hello there!"},
{"voice": "narrator", "text": "[whispering] He is lying."},
))
assert said == ["Hello there! He is lying."]
assert ctrl.history.last().text == "Hello there! He is lying."
def test_a_mid_turn_scene_returns_to_thinking_not_idle(monkeypatch, ctrl):
"""The server is still waiting on the tool result, so the pet talks and
goes back to waiting dropping to IDLE would look like the turn ended."""
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
ctrl._state.transition(PetState.LISTENING)
ctrl._state.transition(PetState.THINKING)
states = _capture(ctrl.state_changed)
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert states == ["talking", "thinking"]
assert ctrl._state.state == PetState.THINKING
def test_the_scene_uses_a_voice_the_server_picked_with_speak_as(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
seen = {}
def capture(inputs, model_id=None, stability=None):
seen["inputs"] = inputs
return np.zeros(4, dtype=np.int16), 24000
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", capture)
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
ctrl._apply_voice(controller_mod.server_client.Reply("ok", "PICKEDvoice123456789", "Terence"))
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert seen["inputs"][0]["voice_id"] == "PICKEDvoice123456789"
def test_a_synthesis_failure_is_reported_back_for_bolt_to_retry(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
def boom(inputs, model_id=None, stability=None):
raise controller_mod.tts.TtsError("voice_id not found")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", boom)
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert "couldn't synthesize" in output and "voice_id not found" in output
assert ctrl._state.state == PetState.IDLE # nothing left half-transitioned
def test_a_bad_voice_name_comes_back_as_advice_not_an_exception(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
output = ctrl._handle_command(_dialogue_command({"voice": "wizard", "text": "hi"}))
assert "unknown voice" in output and "narrator" in output
def test_dialogue_can_be_switched_off_on_this_device(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "DIALOGUE", False)
called = []
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda *a, **k: called.append(1))
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert "disabled" in output and called == []
def test_talking_over_a_scene_is_reported_up_the_relay(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm",
lambda pcm, rate, should_stop=None: False) # barge-in
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
assert "interrupted" in output
# ── petctl self_restart ─────────────────────────────────────────────────────
def test_self_restart_arms_after_the_turn_rather_than_dying_mid_relay(monkeypatch, ctrl, tmp_path):
"""Restarting inline would kill the HTTP tool relay before the result was
posted, and the server would wait out its timeout on a turn that can never
finish. So the command returns, the turn completes, *then* the pet dies."""
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None)
restarts = _capture(ctrl.restart_requested)
output = ctrl._handle_command("petctl self_restart check the new dialogue code")
assert "restarting as soon as this turn finishes" in output
assert restarts == [] # nothing has happened yet
assert ctrl._maybe_self_restart() is True
assert restarts and "check the new dialogue code" in restarts[0]
def test_a_broken_edit_is_reported_instead_of_leaving_nothing_running(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
def boom(*args, **kwargs):
raise controller_mod.self_restart.RestartError(
"the current code does not import, so restarting would leave you with "
"nothing running. Fix this first:\nSyntaxError: invalid syntax"
)
monkeypatch.setattr(controller_mod.self_restart, "preflight", boom)
restarts = _capture(ctrl.restart_requested)
output = ctrl._handle_command("petctl self_restart try the new code")
assert "SyntaxError" in output and "refused" in output
assert ctrl._maybe_self_restart() is False
assert restarts == []
def test_a_second_restart_request_in_one_turn_is_a_no_op(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "ctx.json")
monkeypatch.setattr(controller_mod.self_restart, "preflight", lambda *a, **k: None)
ctrl._handle_command("petctl self_restart first")
assert "already armed" in ctrl._handle_command("petctl self_restart second")
def test_self_restart_can_be_switched_off_on_this_device(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "SELF_RESTART", False)
checked = []
monkeypatch.setattr(controller_mod.self_restart, "preflight",
lambda *a, **k: checked.append(1))
assert "disabled" in ctrl._handle_command("petctl self_restart go")
assert checked == []
def test_coming_back_up_reports_to_the_server_and_speaks_the_reply(monkeypatch, ctrl, tmp_path):
"""The half that makes it a loop: the new process tells Bolt it's back and
why, and his answer is spoken like any other turn."""
state = tmp_path / "ctx.json"
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", state)
controller_mod.self_restart.arm("check the walk cycle", version="0.2.3",
path=state, now=1000.0)
sent, spoken = [], []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or Reply("Good, it's up."))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None:
spoken.append(text) or True)
ctrl._report_self_restart()
assert "[pet self-restart]" in sent[0] and "check the walk cycle" in sent[0]
assert spoken == ["Good, it's up."]
# Consumed, so the next start doesn't announce the same restart again.
assert controller_mod.self_restart.load(state) is None
def test_an_ordinary_start_reports_nothing(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "none.json")
called = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: called.append(text))
ctrl._report_self_restart()
assert called == []
+225
View File
@@ -0,0 +1,225 @@
"""`dialoguectl` — multi-voice scene parsing, voice resolution, API limits,
and the request the ElevenLabs Text to Dialogue endpoint actually gets.
Pure logic plus one mocked HTTP call: no audio device, no network, no display.
"""
import sys
from pathlib import Path
from unittest.mock import MagicMock, patch
import numpy as np
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import dialogue
from bolt_pet.audio import tts
SELF_ID = "aaorr6ZHIL88gEexu7dC"
NARRATOR_ID = "9BWtsMINqrJLrRacOk9x"
VILLAIN_ID = "IKne3meq5aSn9XLyUdCD"
VOICES = {"narrator": NARRATOR_ID, "villain": VILLAIN_ID}
def _scene(*lines):
return '{"lines": [' + ", ".join(lines) + "]}"
# ── parsing ─────────────────────────────────────────────────────────────────
def test_non_dialogue_commands_are_left_alone():
assert dialogue.parse("ls -la") is None
assert dialogue.parse('filectl {"op": "list"}') is None
assert dialogue.parse("") is None
# "dialogues" must not be mistaken for the "dialogue" prefix
assert dialogue.parse("dialogues --list") is None
def test_a_scene_parses_into_lines():
action = dialogue.parse(
'dialoguectl ' + _scene(
'{"voice": "self", "text": "[cheerfully] Hello, how are you?"}',
'{"voice": "villain", "text": "[stuttering] I am... fine."}',
)
)
assert action["action"] == "dialogue"
assert [line["voice"] for line in action["lines"]] == ["self", "villain"]
assert action["lines"][0]["text"].startswith("[cheerfully]")
def test_the_elevenlabs_field_names_are_accepted_too():
"""The model has read that API; copying its shape is the obvious thing to
try, so 'inputs'/'voice_id' work as well as 'lines'/'voice'."""
action = dialogue.parse(
'dialoguectl {"inputs": [{"voice_id": "%s", "text": "hi"}]}' % NARRATOR_ID
)
assert action["lines"] == [{"voice": NARRATOR_ID, "text": "hi"}]
def test_a_line_with_no_voice_defaults_to_the_pet_itself():
action = dialogue.parse('dialoguectl {"lines": [{"text": "just me talking"}]}')
assert action["lines"][0]["voice"] == "self"
def test_truncated_json_explains_the_one_line_rule():
"""The real failure mode: the server's command extractor stops at the
first newline, so a multi-line payload arrives cut in half. The error has
to name the cause, since Bolt is the one who has to fix it."""
with pytest.raises(dialogue.DialogueError, match="one line"):
dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"')
def test_an_empty_or_shapeless_payload_is_rejected():
with pytest.raises(dialogue.DialogueError, match="needs a JSON argument"):
dialogue.parse("dialoguectl")
with pytest.raises(dialogue.DialogueError, match="non-empty"):
dialogue.parse('dialoguectl {"lines": []}')
with pytest.raises(dialogue.DialogueError, match="no text"):
dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": " "}]}')
def test_optional_model_and_stability_ride_along():
action = dialogue.parse(
'dialoguectl {"model_id": "eleven_v3", "stability": 0.8, '
'"lines": [{"text": "hi"}]}'
)
assert action["model"] == "eleven_v3"
assert action["stability"] == 0.8
# ── voice resolution ────────────────────────────────────────────────────────
def test_named_voices_resolve_from_the_configured_cast():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "narrator", "text": "Once upon a time."}',
'{"voice": "villain", "text": "Not this again."}',
))
inputs = dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID)
assert [entry["voice_id"] for entry in inputs] == [NARRATOR_ID, VILLAIN_ID]
def test_self_tracks_the_voice_the_pet_is_currently_using():
"""A scene featuring Bolt should sound like whoever Bolt currently is —
including a voice the server picked mid-conversation with speak_as."""
action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}')
picked = "VOICEfromSPEAKas1234"
assert dialogue.resolve(action, self_voice=picked)[0]["voice_id"] == picked
def test_a_raw_voice_id_passes_straight_through():
action = dialogue.parse('dialoguectl {"lines": [{"voice": "%s", "text": "hi"}]}' % NARRATOR_ID)
assert dialogue.resolve(action, self_voice=SELF_ID)[0]["voice_id"] == NARRATOR_ID
def test_an_unknown_name_lists_what_is_available():
action = dialogue.parse('dialoguectl {"lines": [{"voice": "wizard", "text": "hi"}]}')
with pytest.raises(dialogue.DialogueError) as excinfo:
dialogue.resolve(action, voices=VOICES, self_voice=SELF_ID)
message = str(excinfo.value)
assert "wizard" in message and "narrator" in message and "villain" in message
def test_self_without_a_configured_voice_says_so():
action = dialogue.parse('dialoguectl {"lines": [{"voice": "self", "text": "hi"}]}')
with pytest.raises(dialogue.DialogueError, match="ELEVENLABS_VOICE_ID"):
dialogue.resolve(action, self_voice="")
def test_the_voice_map_parser_skips_typos_instead_of_dying():
voices = dialogue.parse_voice_map(f"narrator:{NARRATOR_ID}, broken-entry, villain:{VILLAIN_ID}")
assert voices == {"narrator": NARRATOR_ID, "villain": VILLAIN_ID}
assert dialogue.parse_voice_map("") == {}
# ── API limits, enforced before the request goes out ────────────────────────
def test_too_many_distinct_voices_is_refused_locally():
inputs = [{"text": "hi", "voice_id": f"voice{index:015d}"} for index in range(11)]
with pytest.raises(dialogue.DialogueError, match="limit is 10"):
dialogue.check_limits(inputs)
def test_an_over_long_scene_is_refused_with_advice():
inputs = [{"text": "x" * 1100, "voice_id": SELF_ID} for _ in range(2)]
with pytest.raises(dialogue.DialogueError) as excinfo:
dialogue.check_limits(inputs)
assert "Split it" in str(excinfo.value) # actionable, since Bolt reads this
# ── display / reporting ─────────────────────────────────────────────────────
def test_delivery_tags_are_stripped_from_what_the_bubble_shows():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "self", "text": "[cheerfully] Hello there!"}',
'{"voice": "narrator", "text": "[whispering] He is lying."}',
))
assert dialogue.spoken_text(action) == "Hello there! He is lying."
def test_the_relay_report_names_the_cast():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "self", "text": "one"}', '{"voice": "narrator", "text": "two"}',
))
assert dialogue.describe(action) == "[dialogue] played 2 lines in 2 voices: narrator, self"
# ── the HTTP request ────────────────────────────────────────────────────────
def _pcm_response(samples=(1, 2, 3, 4)):
response = MagicMock()
response.content = np.array(samples, dtype=np.int16).tobytes()
response.raise_for_status = MagicMock()
return response
def test_the_request_matches_the_text_to_dialogue_api(monkeypatch):
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
monkeypatch.setattr(tts.config, "TTS_SAMPLE_RATE", 24000)
monkeypatch.setattr(tts.config, "DIALOGUE_MODEL_ID", "eleven_v3")
inputs = [
{"text": "[cheerfully] Hello", "voice_id": NARRATOR_ID},
{"text": "[stuttering] H-hi", "voice_id": VILLAIN_ID},
]
with patch.object(tts.requests, "post", return_value=_pcm_response()) as post:
pcm, rate = tts.synthesize_dialogue(inputs)
assert rate == 24000 and pcm.tolist() == [1, 2, 3, 4]
args, kwargs = post.call_args
assert args[0] == "https://api.elevenlabs.io/v1/text-to-dialogue"
assert kwargs["params"] == {"output_format": "pcm_24000"}
assert kwargs["headers"] == {"xi-api-key": "test-key"}
assert kwargs["json"]["inputs"] == inputs
assert kwargs["json"]["model_id"] == "eleven_v3"
assert "settings" not in kwargs["json"] # omitted unless asked for
def test_stability_is_only_sent_when_given(monkeypatch):
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
with patch.object(tts.requests, "post", return_value=_pcm_response()) as post:
tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}], stability=0.3)
assert post.call_args.kwargs["json"]["settings"] == {"stability": 0.3}
def test_a_rejected_request_surfaces_what_the_api_said(monkeypatch):
"""The API explains refusals in the body; Bolt reads this through the tool
relay, so it has to reach him rather than being flattened to '422'."""
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "test-key")
failure = MagicMock()
failure.text = '{"detail": "voice_id not found"}'
error = Exception("422 Client Error")
error.response = failure
response = MagicMock()
response.raise_for_status = MagicMock(side_effect=error)
with patch.object(tts.requests, "post", return_value=response):
with pytest.raises(tts.TtsError, match="voice_id not found"):
tts.synthesize_dialogue([{"text": "hi", "voice_id": "nope"}])
def test_no_api_key_fails_before_the_request(monkeypatch):
monkeypatch.setattr(tts.config, "ELEVENLABS_API_KEY", "")
with patch.object(tts.requests, "post") as post:
with pytest.raises(tts.TtsError):
tts.synthesize_dialogue([{"text": "hi", "voice_id": SELF_ID}])
post.assert_not_called()
+54
View File
@@ -0,0 +1,54 @@
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import file_delivery
def test_sanitize_filename_strips_directory_components():
assert file_delivery.sanitize_filename("../../etc/passwd") == "passwd"
assert file_delivery.sanitize_filename("/absolute/path/report.pdf") == "report.pdf"
assert file_delivery.sanitize_filename("plain.txt") == "plain.txt"
def test_sanitize_filename_falls_back_on_empty_or_dots():
assert file_delivery.sanitize_filename("") == "delivered_file"
assert file_delivery.sanitize_filename("..") == "delivered_file"
assert file_delivery.sanitize_filename(".") == "delivered_file"
assert file_delivery.sanitize_filename(None) == "delivered_file"
def test_unique_path_returns_the_plain_name_when_free(tmp_path):
path = file_delivery.unique_path(tmp_path, "report.pdf")
assert path == tmp_path / "report.pdf"
def test_unique_path_suffixes_on_collision(tmp_path):
(tmp_path / "report.pdf").write_bytes(b"existing")
path = file_delivery.unique_path(tmp_path, "report.pdf")
assert path == tmp_path / "report (1).pdf"
(tmp_path / "report (1).pdf").write_bytes(b"also existing")
path = file_delivery.unique_path(tmp_path, "report.pdf")
assert path == tmp_path / "report (2).pdf"
def test_unique_path_creates_the_directory(tmp_path):
target = tmp_path / "nested" / "dir"
file_delivery.unique_path(target, "a.txt")
assert target.is_dir()
def test_save_writes_bytes_and_sanitizes_the_name(tmp_path):
path = file_delivery.save(tmp_path, "../sneaky/report.pdf", b"hello")
assert path == tmp_path / "report.pdf"
assert path.read_bytes() == b"hello"
def test_save_never_overwrites_an_existing_download(tmp_path):
first = file_delivery.save(tmp_path, "notes.txt", b"first")
second = file_delivery.save(tmp_path, "notes.txt", b"second")
assert first != second
assert first.read_bytes() == b"first"
assert second.read_bytes() == b"second"
+329
View File
@@ -0,0 +1,329 @@
"""filectl parsing + execution — the pseudo-commands the server can relay to
read/write/edit local files instead of a raw shell heredoc.
filectl {"op": ...} is a single-line JSON envelope (not a multi-line
marker block) because it rides the "command" tool marker, which the main
repo's tool-call extractor only captures up to the next newline — see the
module docstring in bolt_pet/file_ops.py."""
import json
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import file_ops
def _cmd(payload: dict) -> str:
return "filectl " + json.dumps(payload)
# ── parsing ──────────────────────────────────────────────────────────────────
def test_non_file_commands_are_left_alone():
assert file_ops.parse("ls -la") is None
assert file_ops.parse("systemctl restart nginx") is None
assert file_ops.parse("") is None
# "filed" must not be mistaken for the "file" prefix
assert file_ops.parse("filed --list") is None
def test_list_parses_defaults():
assert file_ops.parse(_cmd({"op": "list", "path": "/tmp"})) == {
"action": "list", "path": "/tmp", "pattern": "*", "recursive": False,
}
def test_list_parses_pattern_and_recursive():
assert file_ops.parse(_cmd({"op": "list", "path": "/tmp", "pattern": "*.py", "recursive": True})) == {
"action": "list", "path": "/tmp", "pattern": "*.py", "recursive": True,
}
def test_list_requires_a_path():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "list"}))
def test_list_rejects_empty_pattern():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "list", "path": "/tmp", "pattern": ""}))
def test_read_parses_path_and_optional_line_range():
assert file_ops.parse(_cmd({"op": "read", "path": "/tmp/a.txt"})) == {
"action": "read", "path": "/tmp/a.txt", "start": None, "end": None,
}
assert file_ops.parse(_cmd({"op": "read", "path": "/tmp/a.txt", "start": 10, "end": 40})) == {
"action": "read", "path": "/tmp/a.txt", "start": 10, "end": 40,
}
def test_read_requires_a_path():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "read"}))
def test_read_rejects_non_numeric_line_args():
with pytest.raises(file_ops.FileOpError):
file_ops.parse('filectl {"op": "read", "path": "/tmp/a.txt", "start": "start"}')
def test_write_parses_path_and_content():
assert file_ops.parse(_cmd({"op": "write", "path": "/tmp/a.txt", "content": "hello\nworld"})) == {
"action": "write", "path": "/tmp/a.txt", "content": "hello\nworld",
}
def test_write_allows_empty_content():
assert file_ops.parse(_cmd({"op": "write", "path": "/tmp/a.txt", "content": ""})) == {
"action": "write", "path": "/tmp/a.txt", "content": "",
}
def test_write_without_content_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "write", "path": "/tmp/a.txt"}))
def test_edit_parses_old_and_new():
assert file_ops.parse(_cmd({"op": "edit", "path": "/tmp/a.txt", "old": "foo\nbar", "new": "baz"})) == {
"action": "edit", "path": "/tmp/a.txt", "old": "foo\nbar", "new": "baz",
}
def test_edit_rejects_identical_old_and_new():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "edit", "path": "/tmp/a.txt", "old": "same", "new": "same"}))
def test_edit_missing_fields_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "edit", "path": "/tmp/a.txt"}))
def test_content_with_embedded_quotes_and_shell_metacharacters_survives():
payload = 'echo "hi $USER" `whoami` && rm -rf /'
command = _cmd({"op": "write", "path": "/tmp/a.txt", "content": payload})
assert "\n" not in command # stays on one line, as the command marker requires
assert file_ops.parse(command) == {"action": "write", "path": "/tmp/a.txt", "content": payload}
def test_multiline_content_stays_on_one_physical_line():
content = "line one\nline two\nline three with \"quotes\" and \\backslashes\\"
command = _cmd({"op": "write", "path": "/tmp/a.txt", "content": content})
assert "\n" not in command
assert file_ops.parse(command)["content"] == content
def test_help():
assert file_ops.parse("filectl help") == {"action": "help"}
assert file_ops.parse("filectl") == {"action": "help"}
assert file_ops.parse(_cmd({"op": "help"})) == {"action": "help"}
def test_invalid_json_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.parse("filectl {not valid json")
def test_non_object_json_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.parse("filectl [1, 2, 3]")
def test_unknown_op_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.parse(_cmd({"op": "frobnicate", "path": "/tmp/a.txt"}))
# ── execution ────────────────────────────────────────────────────────────────
def test_list_shows_files_and_subdirectories(tmp_path):
(tmp_path / "a.txt").write_text("hi")
(tmp_path / "sub").mkdir()
(tmp_path / "sub" / "b.txt").write_text("nested")
action = file_ops.parse(_cmd({"op": "list", "path": str(tmp_path)}))
output = file_ops.execute(action)
assert "a.txt\t2B" in output
assert "sub/" in output
assert "b.txt" not in output # non-recursive: nested file not shown
def test_list_recursive_finds_nested_files(tmp_path):
(tmp_path / "sub").mkdir()
(tmp_path / "sub" / "b.txt").write_text("nested")
action = file_ops.parse(_cmd({"op": "list", "path": str(tmp_path), "recursive": True}))
output = file_ops.execute(action)
assert "sub/b.txt" in output or "sub\\b.txt" in output # os-dependent separator
def test_list_pattern_filters_entries(tmp_path):
(tmp_path / "a.py").write_text("x")
(tmp_path / "b.txt").write_text("y")
action = file_ops.parse(_cmd({"op": "list", "path": str(tmp_path), "pattern": "*.py"}))
output = file_ops.execute(action)
assert "a.py" in output
assert "b.txt" not in output
def test_list_empty_directory_says_so(tmp_path):
action = file_ops.parse(_cmd({"op": "list", "path": str(tmp_path)}))
output = file_ops.execute(action)
assert "no entries" in output
def test_list_missing_directory_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.execute({"action": "list", "path": "/no/such/dir", "pattern": "*", "recursive": False})
def test_list_rejects_a_file_path(tmp_path):
target = tmp_path / "a.txt"
target.write_text("hi")
with pytest.raises(file_ops.FileOpError):
file_ops.execute({"action": "list", "path": str(target), "pattern": "*", "recursive": False})
def test_list_truncates_past_the_entry_cap(tmp_path, monkeypatch):
monkeypatch.setattr(file_ops, "_MAX_LIST_ENTRIES", 3)
for i in range(5):
(tmp_path / f"f{i}.txt").write_text("x")
action = file_ops.parse(_cmd({"op": "list", "path": str(tmp_path)}))
output = file_ops.execute(action)
assert "truncated" in output
assert output.count(".txt") == 3
def test_write_then_read_round_trips(tmp_path):
target = tmp_path / "notes.txt"
write_action = file_ops.parse(_cmd({"op": "write", "path": str(target), "content": "line one\nline two"}))
result = file_ops.execute(write_action)
assert target.read_text() == "line one\nline two"
assert str(target) in result
read_action = file_ops.parse(_cmd({"op": "read", "path": str(target)}))
output = file_ops.execute(read_action)
assert "line one" in output
assert "line two" in output
def test_read_missing_file_raises():
with pytest.raises(file_ops.FileOpError):
file_ops.execute({"action": "read", "path": "/no/such/file.txt", "start": None, "end": None})
def test_read_respects_line_range(tmp_path):
target = tmp_path / "a.txt"
target.write_text("\n".join(f"line{i}" for i in range(1, 11)))
action = file_ops.parse(_cmd({"op": "read", "path": str(target), "start": 3, "end": 5}))
output = file_ops.execute(action)
assert "line3" in output and "line5" in output
assert "line1" not in output and "line6" not in output
def test_edit_replaces_a_unique_match(tmp_path):
target = tmp_path / "a.py"
target.write_text("def foo():\n return 1\n")
action = file_ops.parse(_cmd({"op": "edit", "path": str(target), "old": "return 1", "new": "return 2"}))
file_ops.execute(action)
assert target.read_text() == "def foo():\n return 2\n"
def test_edit_fails_when_text_not_found(tmp_path):
target = tmp_path / "a.py"
target.write_text("def foo():\n return 1\n")
action = file_ops.parse(_cmd({"op": "edit", "path": str(target), "old": "nope", "new": "x"}))
with pytest.raises(file_ops.FileOpError):
file_ops.execute(action)
assert target.read_text() == "def foo():\n return 1\n" # untouched
def test_edit_fails_when_text_is_ambiguous(tmp_path):
target = tmp_path / "a.py"
target.write_text("x = 1\nx = 1\n")
action = file_ops.parse(_cmd({"op": "edit", "path": str(target), "old": "x = 1", "new": "x = 2"}))
with pytest.raises(file_ops.FileOpError):
file_ops.execute(action)
assert target.read_text() == "x = 1\nx = 1\n" # untouched
def test_write_creates_parent_directories(tmp_path):
target = tmp_path / "nested" / "dir" / "a.txt"
action = file_ops.parse(_cmd({"op": "write", "path": str(target), "content": "hi"}))
file_ops.execute(action)
assert target.read_text() == "hi"
# ── lenient JSON (see relay_json) ───────────────────────────────────────────
# Machine-written JSON fails in a small, repeatable set of ways. Observed live
# 2026-07-30: a stray quote after `false` cost a whole desk turn — rejected,
# re-sent identically, rejected again, then abandoned with a promise to the
# user that nothing fulfilled.
def test_the_stray_quote_that_cost_a_live_turn_now_parses():
action = file_ops.parse(
'filectl {"op":"list","path":"/home/maji/Documents","pattern":"*","recursive":false"}'
)
assert action["action"] == "list"
assert action["path"] == "/home/maji/Documents"
assert action["recursive"] is False
assert action["_repairs"] == ["removed a stray quote after a bare value"]
@pytest.mark.parametrize("payload,expected", [
('{"op": "read", "path": "/tmp/a.txt",}', "removed a trailing comma"),
("{'op': 'read', 'path': '/tmp/a.txt'}", "converted single-quoted strings to double-quoted"),
('{“op”: “read”, “path”: “/tmp/a.txt”}', "replaced smart quotes with straight ones"),
('```json {"op": "read", "path": "/tmp/a.txt"} ```', "stripped a markdown code fence"),
])
def test_common_model_json_mistakes_are_repaired(payload, expected):
action = file_ops.parse("filectl " + payload)
assert action["action"] == "read"
assert action["path"] == "/tmp/a.txt"
assert expected in action["_repairs"]
def test_python_literals_are_converted():
action = file_ops.parse('filectl {"op": "list", "path": "/tmp", "recursive": True}')
assert action["recursive"] is True
def test_a_repair_is_reported_in_the_output_never_hidden(tmp_path):
"""Silently fixing it would work today and guarantee the same broken call
tomorrow the model has to be told while it still has the turn."""
(tmp_path / "a.txt").write_text("hello", encoding="utf-8")
action = file_ops.parse(
'filectl {"op":"list","path":"%s","recursive":false"}' % tmp_path
)
output = file_ops.execute(action)
assert "a.txt" in output
assert "your JSON was malformed" in output
assert "stray quote" in output
def test_valid_json_gets_no_repair_note(tmp_path):
(tmp_path / "a.txt").write_text("hello", encoding="utf-8")
action = file_ops.parse('filectl {"op": "list", "path": "%s"}' % tmp_path)
assert "_repairs" not in action
assert "malformed" not in file_ops.execute(action)
def test_genuinely_unparseable_json_points_at_the_character():
""""Expecting ',' delimiter: char 74" is not something a model can act on."""
with pytest.raises(file_ops.FileOpError) as excinfo:
file_ops.parse('filectl {"op": "read", "path": "/tmp/a.txt" "extra": 1}')
message = str(excinfo.value)
assert "^" in message # caret under the offending character
assert '"extra"' in message # ...and the fragment around it
+112
View File
@@ -0,0 +1,112 @@
"""Local intent recognition — pure string logic, no hardware or display.
The interesting tests are the negative ones: this feature's whole risk is
swallowing something that was meant for the server.
"""
import sys
from pathlib import Path
import pytest
from bolt_pet import intents
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
@pytest.mark.parametrize("said, expected", [
("stop", "stop"),
("Stop!", "stop"),
("never mind", "stop"),
("be quiet", "stop"),
("go to sleep", "nap"),
("take a nap", "nap"),
("goodnight", "nap"),
("wake up", "wake"),
("come here", "come"),
("follow my cursor", "come"),
("get out of the way", "go_away"),
("hide", "go_away"),
("say that again", "repeat"),
("what did you say?", "repeat"),
("go for a walk", "wander_on"),
("stay put", "wander_off"),
("sit", "wander_off"),
("use your normal voice", "voice_reset"),
("go back to your normal voice", "voice_reset"),
("be yourself again", "voice_reset"),
])
def test_recognized_phrases(said, expected):
intent = intents.recognize(said)
assert intent is not None and intent.name == expected
@pytest.mark.parametrize("said", [
# Each of these starts with (or contains) an intent phrase, and every one is
# a real request. A substring match would eat all of them.
"stop the docker container",
"stop the deploy and tell me what broke",
"can you hide the window that's covering my terminal",
"come up with a name for this branch",
"repeat the last command but with sudo",
"what did you say the disk usage was on the server",
"move the config file to the backup directory",
"sit down and write me a haiku about kubernetes",
"what time is it",
"go to sleep mode on the server",
"",
" ",
# Pure filler leaves an empty string, which must not match anything.
"hey bolt",
"okay bolt please",
])
def test_real_requests_are_left_for_the_server(said):
assert intents.recognize(said) is None
def test_filler_is_stripped_from_both_ends():
assert intents.normalize("Hey Bolt, could you please just stop now?") == "stop"
assert intents.normalize("okay, come here buddy") == "come here"
def test_normalize_returns_empty_for_pure_filler():
assert intents.normalize("hey bolt") == ""
assert intents.normalize("...") == ""
def test_intents_carry_ui_actions_in_the_pet_actions_shape():
"""The action dicts go straight to PetWindow.apply_action, so they have to
match the vocabulary pet_actions.parse produces no new UI cases."""
assert intents.recognize("come here").action == {"action": "move", "anchor": "cursor"}
assert intents.recognize("stay put").action == {"action": "wander", "enabled": False}
assert intents.recognize("go to sleep").action == {"action": "nap", "enabled": True}
def test_stop_says_nothing():
"""Answering "okay!" when told to be quiet defeats the purpose."""
intent = intents.recognize("be quiet")
assert intent.speak == "" and intent.action is None
def test_a_phrase_claimed_by_two_intents_fails_at_import(monkeypatch):
"""Without this guard the phrase would silently bind to whichever intent was
declared last a table edit that looks fine and misbehaves on a mic."""
monkeypatch.setattr(intents, "_TABLE", (
(intents.Intent("stop"), ("enough",)),
(intents.Intent("nap"), ("enough",)),
))
with pytest.raises(ValueError, match="claimed by both"):
intents._build()
def test_a_phrase_of_pure_filler_fails_at_import(monkeypatch):
"""It would normalise to "" and then match any all-filler utterance."""
monkeypatch.setattr(intents, "_TABLE", ((intents.Intent("stop"), ("please bolt",)),))
with pytest.raises(ValueError, match="normalises to nothing"):
intents._build()
def test_every_table_phrase_round_trips():
for phrase, intent in intents._BY_PHRASE.items():
assert phrase, "a phrase normalised to nothing"
assert intents.recognize(phrase) is intent
+48
View File
@@ -112,3 +112,51 @@ def test_pcm_to_wav_bytes_round_trips_via_wave_module():
assert wf.getframerate() == 16000
frames = wf.readframes(wf.getnframes())
assert np.frombuffer(frames, dtype=np.int16).tolist() == pcm.tolist()
# ── flushing buffered audio ─────────────────────────────────────────────────
class _BufferedStream:
"""A stream with a backlog, like PortAudio's ring buffer after the reader
was blocked on a network call for a while."""
def __init__(self, available):
self.read_available = available
self.reads = []
def read(self, frames):
self.reads.append(frames)
self.read_available = max(0, self.read_available - frames)
return np.zeros((frames, 1), dtype=np.int16), False
def test_flush_drops_exactly_what_was_buffered():
stream = _BufferedStream(4096)
assert mic.flush(stream) == 4096
assert stream.reads == [4096]
assert stream.read_available == 0
def test_flush_is_bounded_so_it_cannot_chase_a_live_stream():
"""A stream filling as fast as it drains must not spin forever."""
stream = _BufferedStream(10 ** 9)
dropped = mic.flush(stream, max_seconds=1.0, sample_rate=16000)
assert dropped == 16000
def test_flush_is_a_noop_on_an_empty_or_fake_stream():
stream = _BufferedStream(0)
assert mic.flush(stream) == 0
assert stream.reads == []
assert mic.flush(_ScriptedStream([])) == 0 # no read_available at all
assert mic.flush(None) == 0
def test_flush_swallows_a_device_error():
class _Broken:
read_available = 1024
def read(self, frames):
raise RuntimeError("device disappeared")
assert mic.flush(_Broken()) == 0
+169
View File
@@ -0,0 +1,169 @@
"""Screen layout logic — resolving `petctl jump` targets and describing the
setup. Pure: the monitor list is normally published by the UI, so none of this
needs a display."""
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import monitors as m
def grid():
"""The 2x2 setup this was built against: four 1080p screens.
[1 HDMI-0] [2 HDMI-1]
[3 DP-0 ] [4 DP-2 ]
"""
return [
m.Monitor(0, "HDMI-0", 0, 0, 1920, 1080, primary=False),
m.Monitor(1, "HDMI-1", 1920, 0, 1920, 1080, primary=False),
m.Monitor(2, "DP-0", 0, 1080, 1920, 1080, primary=True),
m.Monitor(3, "DP-2", 1920, 1080, 1920, 1080, primary=False),
]
def two():
return [
m.Monitor(0, "eDP-1", 0, 0, 1920, 1080, primary=True),
m.Monitor(1, "HDMI-1", 1920, 0, 2560, 1440),
]
def test_numbers_shown_to_humans_are_one_based():
left, right = two()
assert left.index == 0 and left.number == 1
assert right.index == 1 and right.number == 2
assert "1: eDP-1 1920x1080 (primary)" == left.label
def test_geometry_helpers():
screen = m.Monitor(1, "HDMI-1", 1920, 0, 1920, 1080)
assert screen.right == 3840 and screen.bottom == 1080
assert screen.center == (2880, 540)
assert screen.contains(1920, 0)
assert screen.contains(3839, 1079)
assert not screen.contains(3840, 0) # right edge is exclusive
assert not screen.contains(1919, 0)
def test_monitor_containing_and_nearest():
screens = grid()
assert m.monitor_containing(screens, 100, 100).name == "HDMI-0"
assert m.monitor_containing(screens, 2000, 1500).name == "DP-2"
assert m.monitor_containing(screens, -50, -50) is None
# off the desktop entirely still resolves to something
assert m.nearest_monitor(screens, -500, -500).name == "HDMI-0"
def test_resolve_by_number():
screens = grid()
assert m.resolve(screens, "3").name == "DP-0"
with pytest.raises(ValueError, match="no monitor 9"):
m.resolve(screens, "9")
with pytest.raises(ValueError):
m.resolve(screens, "0")
def test_resolve_next_and_prev_wrap():
screens = grid()
assert m.resolve(screens, "next", current=3).number == 1
assert m.resolve(screens, "prev", current=0).number == 4
assert m.resolve(screens, "next", current=0).number == 2
def test_resolve_primary_and_other():
screens = grid()
assert m.resolve(screens, "primary", current=0).name == "DP-0"
# "other" on a two-screen setup is genuinely the other one
pair = two()
assert m.resolve(pair, "other", current=0).number == 2
assert m.resolve(pair, "other", current=1).number == 1
def test_resolve_directions_on_a_grid():
screens = grid()
# from top-left (HDMI-0)
assert m.resolve(screens, "right", current=0).name == "HDMI-1"
assert m.resolve(screens, "down", current=0).name == "DP-0"
# from bottom-right (DP-2)
assert m.resolve(screens, "left", current=3).name == "DP-0"
assert m.resolve(screens, "up", current=3).name == "HDMI-1"
def test_direction_prefers_the_best_aligned_screen():
screens = grid()
# "right" from DP-0 (bottom-left) must pick DP-2 (same row), not HDMI-1,
# even though both are to the right.
assert m.resolve(screens, "right", current=2).name == "DP-2"
def test_resolve_direction_with_nothing_there():
screens = grid()
with pytest.raises(ValueError, match="no monitor to the left"):
m.resolve(screens, "left", current=0)
def test_resolve_by_name_is_fuzzy_but_refuses_ambiguity():
screens = grid()
assert m.resolve(screens, "dp-2").name == "DP-2"
assert m.resolve(screens, "HDMI-0").name == "HDMI-0"
with pytest.raises(ValueError, match="matches several"):
m.resolve(screens, "hdmi")
def test_resolve_unknown_spec_lists_the_options():
screens = two()
with pytest.raises(ValueError) as excinfo:
m.resolve(screens, "the big one")
assert "eDP-1" in str(excinfo.value) and "HDMI-1" in str(excinfo.value)
def test_resolve_without_a_current_screen_falls_back_to_primary():
screens = grid()
# primary is index 2, so "next" from nowhere is index 3
assert m.resolve(screens, "next", current=None).number == 4
# an out-of-range current is treated the same way rather than exploding
assert m.resolve(screens, "next", current=99).number == 4
def test_resolve_needs_monitors():
with pytest.raises(ValueError, match="no monitors"):
m.resolve([], "next")
with pytest.raises(ValueError, match="needs a target"):
m.resolve(grid(), "")
def test_random_always_moves_somewhere_else():
screens = grid()
for current in range(4):
assert m.resolve(screens, "random", current=current).index != current
def test_summary_is_one_line_and_marks_where_the_pet_is():
line = m.summary(grid(), current=1)
assert "\n" not in line
assert line.startswith("4 monitors:")
assert "Bolt is on 2" in line
assert m.summary([]) is None
assert "Bolt is on" not in m.summary(grid(), current=None)
def test_annotate_matches_the_screen_context_style():
out = m.annotate("what's on the other screen?", grid(), current=0)
assert out.startswith("what's on the other screen?")
assert "[4 monitors:" in out
# nothing to say, nothing added
assert m.annotate("hello", [], None) == "hello"
assert m.annotate("", grid(), 0) == ""
def test_describe_lists_every_screen_and_flags_the_pet():
text = m.describe(grid(), current=2)
assert text.count("\n") == 4 # header + 4 screens
assert "DP-0" in text and "+0+1080" in text
assert text.count("Bolt is here") == 1
assert m.describe([]) == "[pet] no monitor information available"
+14
View File
@@ -69,3 +69,17 @@ def test_describe_is_reported_back_to_the_server():
assert "top-left" in pet_actions.describe({"action": "move", "anchor": "top-left"})
assert "wave" in pet_actions.describe({"action": "emote", "emote": "wave"})
assert pet_actions.describe({"action": "help"}) == pet_actions.HELP
def test_voice_reset_parses_with_or_without_the_word_reset():
assert pet_actions.parse("petctl voice reset") == {"action": "voice", "voice": "default"}
assert pet_actions.parse("petctl voice default") == {"action": "voice", "voice": "default"}
assert pet_actions.parse("petctl voice") == {"action": "voice", "voice": "default"}
def test_petctl_cannot_be_used_to_pick_a_voice():
"""Choosing a voice is the server's job (speak_as) — it has the voice
library. petctl only ever undoes one, so an attempt to set a voice here
is pointed back at the marker that works."""
with pytest.raises(pet_actions.ActionError, match="speak_as"):
pet_actions.parse("petctl voice Terence")
+154 -1
View File
@@ -15,7 +15,10 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import config
from bolt_pet.state import PetState
from bolt_pet.ui.pet_window import _EMOTE_TICKS, PetWindow, emote_transform
from bolt_pet.ui.pet_window import (
_EMOTE_TICKS, _WALK_PIXELS_PER_FRAME, PetWindow, emote_transform,
)
from bolt_pet.ui.sprite import WALK
@pytest.fixture(scope="module")
@@ -179,3 +182,153 @@ def test_click_through_toggles_mouse_transparency(pet):
pet.set_click_through(False)
assert pet.testAttribute(Qt.WA_TransparentForMouseEvents) is False
# ── monitors ────────────────────────────────────────────────────────────────
def test_window_publishes_a_monitor_list(pet):
"""Whatever the test host's screen setup is, the window must describe it
in the shape the controller expects."""
monitors = pet.monitors()
assert monitors, "offscreen Qt still reports at least one screen"
assert [m.index for m in monitors] == list(range(len(monitors)))
assert all(m.width > 0 and m.height > 0 for m in monitors)
assert all(m.name for m in monitors)
assert sum(1 for m in monitors if m.primary) <= 1
def test_publish_monitors_re_emits_when_forced(pet):
"""ui/app.py relies on this: the window is built before the controller
exists, so its constructor's publish reaches nobody and has to be redone."""
seen = []
pet.monitors_changed.connect(seen.append)
pet.publish_monitors() # force defaults to True
assert len(seen) == 1
pet.publish_monitors(force=False) # nothing changed -> stays quiet
assert len(seen) == 1
def test_pet_reports_which_monitor_it_is_on(pet):
seen = []
pet.pet_monitor_changed.connect(seen.append)
pet.publish_monitors()
assert seen and seen[-1] == pet.current_monitor_index()
assert 0 <= seen[-1] < len(pet.monitors())
def test_jump_moves_the_window_onto_the_target_screen(pet):
monitors = pet.monitors()
target = len(monitors) - 1
pet.apply_action({"action": "jump", "monitor": target})
assert pet.current_monitor_index() == target
# a jump lands with a hop rather than sliding there
assert pet._emote == "hop"
def test_jump_cancels_a_stroll_so_it_does_not_walk_back(pet):
pet.apply_action({"action": "move", "anchor": "top-left"})
assert pet._wander_target is not None
pet.apply_action({"action": "jump", "monitor": 0})
assert pet._wander_target is None
assert pet._commanded_move is False
def test_jump_to_a_bogus_index_is_a_no_op(pet):
before = pet.pos()
pet.apply_action({"action": "jump", "monitor": 99})
pet.apply_action({"action": "jump", "monitor": -1})
assert pet.pos() == before
# ── walk cycle ──────────────────────────────────────────────────────────────
def test_walk_art_loads_as_a_non_state_animation(pet):
"""Walking is a property of movement, not a PetState, so it lives outside
the state machine but still loads like any other animation."""
assert pet.sprites.has(WALK)
assert len(pet.sprites.get(WALK).frames) == 8
assert pet.sprites.get("nonsense") is pet.sprites.get(PetState.IDLE)
assert not pet.sprites.has("nonsense")
def test_walking_overrides_the_state_animation(pet):
assert pet._animation_key() == pet._current_state
pet._advance_walk(50, 0, 1.0)
assert pet._animation_key() == WALK
def test_walk_cycle_advances_by_distance_not_by_the_clock(pet):
"""The planted paw tracks backwards at the speed the window moves
forwards; drive it off the animation timer instead and the feet skate."""
anim = pet.sprites.get(WALK)
anim.reset()
pet._advance_walk(100, 0, _WALK_PIXELS_PER_FRAME * 3)
assert anim._index == 3
# a step too small to cross the threshold banks the distance instead
pet._advance_walk(100, 0, _WALK_PIXELS_PER_FRAME * 0.5)
assert anim._index == 3
pet._advance_walk(100, 0, _WALK_PIXELS_PER_FRAME * 0.5)
assert anim._index == 4
def test_the_animation_timer_does_not_double_step_the_walk(pet):
anim = pet.sprites.get(WALK)
pet._advance_walk(50, 0, 1.0)
anim.reset()
pet._advance_frame()
assert anim._index == 0
def test_facing_follows_horizontal_travel(pet):
pet._advance_walk(50, 0, 1.0)
assert pet._facing == 1
pet._advance_walk(-50, 0, 1.0)
assert pet._facing == -1
def test_a_near_vertical_stroll_does_not_flip_him(pet):
"""Rounding noise on dx would otherwise flip him back and forth every
tick on a straight-up walk."""
pet._facing = 1
pet._advance_walk(0.4, 60, 1.0)
assert pet._facing == 1
def test_walking_left_paints_a_mirrored_frame(pet):
frame = pet.sprites.get(WALK).current()
pet._facing = 1
assert pet._oriented(frame) is frame # art is drawn facing right
pet._facing = -1
flipped = pet._oriented(frame)
assert flipped is not frame
assert flipped.size() == frame.size()
assert pet._oriented(frame) is flipped # cached, not re-flipped per paint
def test_stopping_resets_the_cycle_to_a_standing_frame(pet):
pet._advance_walk(50, 0, _WALK_PIXELS_PER_FRAME * 2)
assert pet._walking
pet._stop_walking()
assert not pet._walking
assert pet._walk_distance == 0.0
assert pet.sprites.get(WALK)._index == 0
assert pet._animation_key() == pet._current_state
def test_walk_art_suppresses_the_hard_coded_bob(pet):
"""The frames carry their own weight shift — bobbing the window as well
would double it up."""
pet._advance_walk(50, 0, 5.0)
assert pet._bob_offset == 0
def test_without_walk_art_it_falls_back_to_the_old_bob(qt_app, tmp_path):
window = PetWindow(sprite_dir=tmp_path)
try:
assert not window.sprites.has(WALK)
window._advance_walk(50, 0, 5.0)
assert window._walking
assert window._animation_key() == window._current_state
assert window._bob_offset < 0 # still visibly moving
finally:
window.close()
+168
View File
@@ -0,0 +1,168 @@
"""OCR plumbing for `petctl read` — engine selection and output cleanup.
Only the pure half is covered, per the testing conventions: capture and the
OCR call itself need a real screen and a real engine. Engine probes are
injected so these pass on a machine with a different set installed (or none).
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import screen_text
from bolt_pet.monitors import Monitor
def probes(modules=(), binaries=()):
return (lambda name: name in modules), (
lambda name: f"/usr/bin/{name}" if name in binaries else None
)
def test_prefers_tesseract_when_fully_installed():
has_module, which = probes({"pytesseract"}, {"tesseract"})
assert screen_text.resolve_engine(has_module, which) == ("pytesseract", "")
def test_falls_back_to_rapidocr_when_tesseract_binary_is_absent():
has_module, which = probes({"pytesseract", "rapidocr_onnxruntime"}, set())
engine, reason = screen_text.resolve_engine(has_module, which)
assert engine == "rapidocr"
assert reason == ""
def test_pytesseract_without_the_binary_says_which_half_is_missing():
"""The commonest broken setup: `pip install pytesseract` and stop, not
realising the actual engine is a system package."""
has_module, which = probes({"pytesseract"}, set())
engine, reason = screen_text.resolve_engine(has_module, which)
assert engine is None
assert "tesseract binary" in reason
assert "apt install tesseract-ocr" in reason
def test_nothing_installed_explains_how_to_fix_it():
has_module, which = probes(set(), set())
engine, reason = screen_text.resolve_engine(has_module, which)
assert engine is None
assert "pip install" in reason
def test_capture_availability_follows_mss():
assert screen_text.capture_available(lambda name: name == "mss")
assert not screen_text.capture_available(lambda name: False)
def test_clean_drops_ocr_noise_and_blank_runs():
raw = "Firefox\n\n\n |\n .\nBuild failed\n~\n"
assert screen_text.clean_ocr_text(raw) == "Firefox\nBuild failed"
def test_clean_collapses_whitespace_but_keeps_line_structure():
raw = " File edit view \nline\ttwo "
assert screen_text.clean_ocr_text(raw) == "File edit view\nline two"
def test_clean_drops_consecutive_duplicates_only():
raw = "Terminal\nTerminal\nEditor\nTerminal"
assert screen_text.clean_ocr_text(raw) == "Terminal\nEditor\nTerminal"
def test_clean_keeps_short_but_real_tokens():
# two alphanumerics is the bar — "ok" and "42" survive, "-" doesn't
assert screen_text.clean_ocr_text("ok\n-\n42") == "ok\n42"
def test_clean_truncates_and_says_so():
out = screen_text.clean_ocr_text("word " * 500, max_chars=100)
assert out.endswith("[truncated]")
# the cap applies to the text, before the marker is appended
assert len(out.split("\n[truncated]")[0]) <= 100
def test_clean_handles_empty_input():
assert screen_text.clean_ocr_text("") == ""
assert screen_text.clean_ocr_text(None) == ""
def test_format_reading_names_the_monitor():
monitor = Monitor(1, "HDMI-1", 1920, 0, 1920, 1080)
out = screen_text.format_reading(monitor, "Build failed")
assert "monitor 2 (HDMI-1)" in out
assert out.endswith("Build failed")
def test_format_reading_when_nothing_was_recognised():
monitor = Monitor(0, "DP-0", 0, 0, 1920, 1080)
assert "no text recognised" in screen_text.format_reading(monitor, " ")
def test_read_monitor_never_raises_without_an_engine(monkeypatch):
"""Its return value goes back to the server as command output, so every
failure has to come back as a sentence rather than an exception."""
monkeypatch.setattr(screen_text, "capture_available", lambda *a, **k: True)
monkeypatch.setattr(
screen_text, "resolve_engine", lambda *a, **k: (None, "no engine here")
)
out = screen_text.read_monitor(Monitor(0, "DP-0", 0, 0, 1920, 1080))
assert out.startswith("[pet]")
assert "no engine here" in out
def test_read_monitor_reports_a_failed_capture(monkeypatch):
monkeypatch.setattr(screen_text, "capture_available", lambda *a, **k: True)
monkeypatch.setattr(screen_text, "resolve_engine", lambda *a, **k: ("pytesseract", ""))
monkeypatch.setattr(screen_text, "capture", lambda monitor: None)
out = screen_text.read_monitor(Monitor(0, "DP-0", 0, 0, 1920, 1080))
assert "couldn't capture" in out and "Wayland" in out
def test_read_monitor_survives_an_exploding_engine(monkeypatch):
monkeypatch.setattr(screen_text, "capture_available", lambda *a, **k: True)
monkeypatch.setattr(screen_text, "resolve_engine", lambda *a, **k: ("pytesseract", ""))
monkeypatch.setattr(screen_text, "capture", lambda monitor: object())
monkeypatch.setattr(
screen_text, "_ocr", lambda image, engine: (_ for _ in ()).throw(RuntimeError("boom"))
)
out = screen_text.read_monitor(Monitor(0, "DP-0", 0, 0, 1920, 1080))
assert "OCR failed" in out and "boom" in out
def test_read_monitors_splits_the_budget(monkeypatch):
seen = []
def fake_read(monitor, max_chars):
seen.append((monitor.number, max_chars))
return f"screen {monitor.number}"
monkeypatch.setattr(screen_text, "read_monitor", fake_read)
screens = [
Monitor(0, "A", 0, 0, 100, 100),
Monitor(1, "B", 100, 0, 100, 100),
Monitor(2, "C", 200, 0, 100, 100),
]
out = screen_text.read_monitors(screens, 3000)
assert [n for n, _ in seen] == [1, 2, 3]
assert all(limit == 1000 for _, limit in seen)
assert out.count("screen ") == 3
def test_read_monitors_keeps_a_floor_on_the_budget(monkeypatch):
monkeypatch.setattr(
screen_text, "read_monitor", lambda monitor, max_chars: str(max_chars)
)
screens = [Monitor(i, str(i), 0, 0, 10, 10) for i in range(20)]
# 100/20 would be 5 characters per screen, which is useless — floor wins
assert "400" in screen_text.read_monitors(screens, 100)
def test_read_monitors_with_one_screen_uses_the_whole_budget(monkeypatch):
monkeypatch.setattr(
screen_text, "read_monitor", lambda monitor, max_chars: str(max_chars)
)
assert screen_text.read_monitors([Monitor(0, "A", 0, 0, 10, 10)], 4000) == "4000"
def test_read_monitors_with_no_screens():
assert "no monitor information" in screen_text.read_monitors([], 4000)
+233
View File
@@ -0,0 +1,233 @@
"""End-to-end wiring for the multi-monitor features: `petctl jump`, `petctl
monitors`, `petctl read`, and the screen-layout note that rides along with
each utterance.
Needs a QApplication (signals), so run with QT_QPA_PLATFORM=offscreen.
"""
import sys
from pathlib import Path
import pytest
from PySide6.QtWidgets import QApplication
from bolt_pet import controller as controller_mod
from bolt_pet import pet_actions
from bolt_pet.monitors import Monitor
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
_app = QApplication.instance() or QApplication(["test"])
@pytest.fixture(autouse=True)
def no_screen_probes(monkeypatch):
monkeypatch.setattr(controller_mod.screen_context, "is_fullscreen_active", lambda: False)
monkeypatch.setattr(controller_mod.screen_context, "context_for", lambda text: text)
@pytest.fixture
def ctrl():
controller = controller_mod.PetController()
controller.set_monitors(
[
Monitor(0, "HDMI-0", 0, 0, 1920, 1080),
Monitor(1, "HDMI-1", 1920, 0, 1920, 1080),
Monitor(2, "DP-0", 0, 1080, 1920, 1080, primary=True),
]
)
controller.set_pet_monitor(0)
return controller
def _capture(signal):
events = []
signal.connect(lambda *a: events.append(a[0] if len(a) == 1 else a))
return events
# ── parsing ─────────────────────────────────────────────────────────────────
def test_jump_parses_without_validating_the_target():
"""Which monitors exist is a runtime fact, so the pure parser passes the
spec through and monitors.resolve() judges it later."""
assert pet_actions.parse("petctl jump 2") == {"action": "jump", "target": "2"}
assert pet_actions.parse("petctl jump next") == {"action": "jump", "target": "next"}
assert pet_actions.parse("petctl monitor left") == {"action": "jump", "target": "left"}
assert pet_actions.parse("petctl screen HDMI-1") == {"action": "jump", "target": "HDMI-1"}
# a nonsense target is still parsed — it fails at resolve time, with a
# message listing the real monitors
assert pet_actions.parse("petctl jump sideways") == {
"action": "jump", "target": "sideways",
}
def test_jump_needs_a_target():
with pytest.raises(pet_actions.ActionError, match="needs a monitor"):
pet_actions.parse("petctl jump")
def test_read_defaults_to_the_current_screen():
assert pet_actions.parse("petctl read") == {"action": "read", "target": "here"}
assert pet_actions.parse("petctl read all") == {"action": "read", "target": "all"}
assert pet_actions.parse("petctl look 2") == {"action": "read", "target": "2"}
assert pet_actions.parse("petctl see here") == {"action": "read", "target": "here"}
def test_monitors_verb():
for spelling in ("monitors", "screens", "displays"):
assert pet_actions.parse(f"petctl {spelling}") == {"action": "monitors"}
def test_help_mentions_the_new_verbs():
assert "petctl jump" in pet_actions.HELP
assert "petctl read" in pet_actions.HELP
assert "petctl monitors" in pet_actions.HELP
# ── jump ────────────────────────────────────────────────────────────────────
def test_jump_resolves_to_an_index_before_reaching_the_window(ctrl):
"""The controller resolves and emits a concrete index, so the window can't
re-resolve the spec against a different screen ordering."""
actions = _capture(ctrl.action)
out = ctrl._handle_command("petctl jump next")
assert actions == [{"action": "jump", "monitor": 1}]
assert "monitor 2: HDMI-1" in out
def test_jump_by_direction_uses_the_published_layout(ctrl):
actions = _capture(ctrl.action)
ctrl._handle_command("petctl jump down")
assert actions == [{"action": "jump", "monitor": 2}]
def test_jump_tracks_where_the_pet_actually_is(ctrl):
ctrl.set_pet_monitor(1)
actions = _capture(ctrl.action)
ctrl._handle_command("petctl jump left")
assert actions == [{"action": "jump", "monitor": 0}]
def test_jump_to_a_nonexistent_monitor_reports_back_and_moves_nothing(ctrl):
actions = _capture(ctrl.action)
out = ctrl._handle_command("petctl jump 7")
assert actions == []
assert "no monitor 7" in out
assert "you have 3" in out
def test_jump_never_reaches_the_shell(monkeypatch, ctrl):
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", ran.append)
ctrl._handle_command("petctl jump 2")
ctrl._handle_command("petctl monitors")
ctrl._handle_command("petctl read")
assert ran == []
# ── monitors ────────────────────────────────────────────────────────────────
def test_monitors_lists_the_layout_and_where_the_pet_is(ctrl):
ctrl.set_pet_monitor(2)
out = ctrl._handle_command("petctl monitors")
assert "3 monitor(s)" in out
assert "HDMI-0" in out and "DP-0" in out
assert out.count("Bolt is here") == 1
def test_monitors_before_the_ui_has_published_anything():
fresh = controller_mod.PetController()
assert "no monitor information" in fresh._handle_command("petctl monitors")
# ── read ────────────────────────────────────────────────────────────────────
def test_read_here_uses_the_pets_own_screen(monkeypatch, ctrl):
seen = []
monkeypatch.setattr(
controller_mod.screen_text, "read_monitor",
lambda monitor, limit: seen.append(monitor.number) or "text",
)
ctrl.set_pet_monitor(1)
assert ctrl._handle_command("petctl read") == "text"
assert seen == [2]
def test_read_a_named_screen(monkeypatch, ctrl):
seen = []
monkeypatch.setattr(
controller_mod.screen_text, "read_monitor",
lambda monitor, limit: seen.append(monitor.name) or "text",
)
ctrl._handle_command("petctl read DP-0")
assert seen == ["DP-0"]
def test_read_all_goes_through_the_multi_screen_path(monkeypatch, ctrl):
seen = []
monkeypatch.setattr(
controller_mod.screen_text, "read_monitors",
lambda monitors, limit: seen.append(len(monitors)) or "everything",
)
assert ctrl._handle_command("petctl read all") == "everything"
assert seen == [3]
def test_read_an_unknown_screen_explains_rather_than_raising(ctrl):
out = ctrl._handle_command("petctl read 9")
assert out.startswith("[pet]")
assert "no monitor 9" in out
def test_read_respects_the_kill_switch(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "SCREEN_TEXT", False)
called = []
monkeypatch.setattr(
controller_mod.screen_text, "read_monitor",
lambda *a, **k: called.append(1) or "text",
)
out = ctrl._handle_command("petctl read")
assert called == []
assert "disabled" in out and "SCREEN_TEXT" in out
def test_read_passes_the_character_cap_through(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "SCREEN_TEXT_MAX_CHARS", 123)
seen = []
monkeypatch.setattr(
controller_mod.screen_text, "read_monitor",
lambda monitor, limit: seen.append(limit) or "text",
)
ctrl._handle_command("petctl read")
assert seen == [123]
# ── per-turn context ────────────────────────────────────────────────────────
def test_layout_rides_along_with_each_utterance(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "MONITOR_CONTEXT", True)
out = ctrl._with_context("what's on the other screen?")
assert out.startswith("what's on the other screen?")
assert "3 monitors:" in out
assert "Bolt is on 1" in out
def test_layout_context_can_be_switched_off(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "MONITOR_CONTEXT", False)
assert ctrl._with_context("hello") == "hello"
def test_screen_text_never_rides_along_automatically(monkeypatch, ctrl):
"""The layout is free; the *contents* cost an OCR pass and a lot of
privacy, so they only ever move on an explicit petctl read."""
monkeypatch.setattr(controller_mod.config, "MONITOR_CONTEXT", True)
called = []
monkeypatch.setattr(
controller_mod.screen_text, "read_monitor", lambda *a, **k: called.append(1)
)
monkeypatch.setattr(
controller_mod.screen_text, "read_monitors", lambda *a, **k: called.append(1)
)
ctrl._with_context("hello")
assert called == []
+158
View File
@@ -0,0 +1,158 @@
"""`petctl self_restart` — the pet restarting itself and remembering why.
Everything here runs against a temp context file and a fake subprocess runner,
so the tests exercise the arming/preflight/report logic without any process
actually dying.
"""
import subprocess
import sys
from pathlib import Path
from types import SimpleNamespace
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import pet_actions, self_restart
@pytest.fixture
def state(tmp_path):
return tmp_path / "restart_context.json"
def _ok_run(*args, **kwargs):
return SimpleNamespace(returncode=0, stdout="", stderr="")
def _broken_run(*args, **kwargs):
return SimpleNamespace(
returncode=1, stdout="",
stderr=' File "bolt_pet/controller.py", line 42\n def _speak(\nSyntaxError: invalid syntax',
)
# ── parsing ─────────────────────────────────────────────────────────────────
def test_self_restart_parses_with_a_free_text_reason():
action = pet_actions.parse("petctl self_restart check the new walk cycle loads")
assert action == {
"action": "self_restart", "reason": "check the new walk cycle loads",
}
def test_self_restart_needs_no_reason_and_accepts_aliases():
assert pet_actions.parse("petctl self_restart")["reason"] == ""
assert pet_actions.parse("petctl restart")["action"] == "self_restart"
assert pet_actions.parse("petctl reboot")["action"] == "self_restart"
def test_self_restart_is_listed_in_the_help():
assert "self_restart" in pet_actions.HELP
# ── preflight ───────────────────────────────────────────────────────────────
def test_preflight_passes_when_the_code_imports():
self_restart.preflight(run=_ok_run) # no exception
def test_preflight_hands_back_the_traceback_instead_of_dying(state):
"""The whole point: a syntax error Bolt just introduced comes back as
something he can read and fix, in the same turn, with the pet still up."""
with pytest.raises(self_restart.RestartError) as excinfo:
self_restart.preflight(run=_broken_run)
message = str(excinfo.value)
assert "does not import" in message
assert "SyntaxError" in message and "controller.py" in message
def test_preflight_runs_the_import_in_a_subprocess_not_here():
"""This process holds the *old* modules, so an in-process import would
pass on a file that no longer parses."""
seen = {}
def capture(cmd, **kwargs):
seen["cmd"], seen["kwargs"] = cmd, kwargs
return SimpleNamespace(returncode=0, stdout="", stderr="")
self_restart.preflight(run=capture)
assert seen["cmd"][0] == sys.executable
assert "import bolt_pet" in seen["cmd"][2]
assert seen["kwargs"]["env"]["QT_QPA_PLATFORM"] == "offscreen" # imports need no display
def test_a_subprocess_that_cannot_even_run_is_reported(monkeypatch):
def explode(*args, **kwargs):
raise OSError("no python here")
with pytest.raises(self_restart.RestartError, match="couldn't run the preflight"):
self_restart.preflight(run=explode)
# ── context across the restart ──────────────────────────────────────────────
def test_arming_persists_the_reason_for_the_next_process(state):
self_restart.arm("check the sprite frames load", version="0.2.3",
session="pet-desktop", recent=["you: reload the sprites"],
path=state, now=1000.0)
revived = self_restart.load(state)
assert revived.reason == "check the sprite frames load"
assert revived.version == "0.2.3"
assert revived.recent == ["you: reload the sprites"]
assert revived.restarts == [1000.0]
def test_no_context_means_a_normal_start(state):
assert self_restart.load(state) is None
def test_a_corrupt_context_file_is_ignored_not_fatal(state):
state.write_text("{not json at all", encoding="utf-8")
assert self_restart.load(state) is None
def test_clearing_the_context_stops_it_being_re_announced(state):
self_restart.arm("once", path=state, now=1000.0)
self_restart.clear(state)
assert self_restart.load(state) is None
self_restart.clear(state) # clearing twice is not an error
def test_the_report_says_what_happened_and_what_to_check(state):
context = self_restart.arm(
"verify the dialogue command works", verify="verify the dialogue command works",
version="0.2.3", recent=["you: try a scene"], path=state, now=1000.0,
)
text = self_restart.report(context, version="0.2.4", now=1004.5)
assert "I restarted myself" in text
assert "verify the dialogue command works" in text
assert "4.5s" in text
assert "0.2.4" in text and "was 0.2.3" in text
assert "you: try a scene" in text
# ── loop guard ──────────────────────────────────────────────────────────────
def test_restart_history_accumulates_across_restarts(state):
self_restart.arm("one", path=state, now=1000.0)
self_restart.arm("two", path=state, now=1100.0)
assert self_restart.load(state).restarts == [1000.0, 1100.0]
def test_too_many_restarts_in_the_window_is_refused(state):
now = 1000.0
for index in range(self_restart.MAX_RESTARTS):
self_restart.arm(f"attempt {index}", path=state, now=now + index)
with pytest.raises(self_restart.RestartError, match="looping"):
self_restart.check_loop_guard(self_restart.load(state), now=now + 10)
def test_old_restarts_fall_out_of_the_window(state):
now = 1000.0
for index in range(self_restart.MAX_RESTARTS):
self_restart.arm(f"attempt {index}", path=state, now=now + index)
later = now + self_restart.WINDOW_SECONDS + 60
self_restart.check_loop_guard(self_restart.load(state), now=later) # no exception
assert self_restart.recent_restarts(self_restart.load(state), now=later) == []
+111 -2
View File
@@ -1,4 +1,6 @@
import os
import sys
import time
from pathlib import Path
from unittest.mock import MagicMock, patch
@@ -27,7 +29,8 @@ def test_converse_returns_reply_directly():
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response({"type": "reply", "text": "hello there"})
result = server_client.converse("hi")
assert result == "hello there"
assert result.text == "hello there"
assert result.voice_id == "" # no speak_as on this reply
post.assert_called_once()
args, kwargs = post.call_args
assert args[0] == "http://test-server:5002/desk/converse"
@@ -43,7 +46,7 @@ def test_converse_relays_a_command_then_returns_reply():
with patch.object(server_client.requests, "post", side_effect=responses) as post:
on_command = MagicMock(return_value="[exit 0]\nhi")
result = server_client.converse("run echo hi", on_command=on_command)
assert result == "done"
assert result.text == "done"
on_command.assert_called_once_with("echo hi")
# second call was to /desk/tool_result with the command's output
second_call = post.call_args_list[1]
@@ -53,6 +56,19 @@ def test_converse_relays_a_command_then_returns_reply():
}
def test_converse_carries_a_speak_as_voice_back_with_the_reply():
"""The server tags a reply with the voice it picked (speak_as); this
client is what actually speaks in it, so the id has to survive the
return trip rather than being dropped with the rest of the payload."""
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response({
"type": "reply", "text": "Ahoy there.",
"voice_id": "hnhGxwvHP8fc469w51rM", "voice_name": "Terence",
})
result = server_client.converse("talk like a pirate")
assert result == server_client.Reply("Ahoy there.", "hnhGxwvHP8fc469w51rM", "Terence")
def test_converse_raises_server_error_on_error_payload():
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response({"type": "error", "error": "unauthorized"})
@@ -84,3 +100,96 @@ def test_check_health_returns_parsed_json():
get.return_value = _mock_response({"ok": True, "service": "bolt-desk-api"})
result = server_client.check_health()
assert result == {"ok": True, "service": "bolt-desk-api"}
def test_list_outbox_files_returns_the_queue():
with patch.object(server_client.requests, "get") as get:
get.return_value = _mock_response(
{"files": [{"id": "abc", "name": "report.pdf", "size": 9}]}
)
result = server_client.list_outbox_files()
assert result == [{"id": "abc", "name": "report.pdf", "size": 9}]
args, kwargs = get.call_args
assert args[0] == "http://test-server:5002/desk/files"
assert kwargs["params"] == {"session_id": "pet-test"}
assert kwargs["headers"] == {"X-Desk-Api-Key": "test-key"}
def test_list_outbox_files_defaults_to_empty_list():
with patch.object(server_client.requests, "get") as get:
get.return_value = _mock_response({})
assert server_client.list_outbox_files() == []
def test_list_outbox_files_raises_server_error_when_unreachable():
with patch.object(server_client.requests, "get", side_effect=ConnectionError("no route")):
with pytest.raises(server_client.ServerError):
server_client.list_outbox_files()
def test_download_outbox_file_returns_raw_bytes():
with patch.object(server_client.requests, "get") as get:
resp = _mock_response({})
resp.content = b"%PDF fake bytes"
get.return_value = resp
result = server_client.download_outbox_file("abc")
assert result == b"%PDF fake bytes"
args, kwargs = get.call_args
assert args[0] == "http://test-server:5002/desk/files/abc"
assert kwargs["params"] == {"session_id": "pet-test"}
def test_download_outbox_file_raises_server_error_on_http_failure():
with patch.object(server_client.requests, "get") as get:
get.return_value = _mock_response({}, ok=False)
with pytest.raises(server_client.ServerError, match="abc"):
server_client.download_outbox_file("abc")
def test_converse_reports_a_relay_that_never_produced_a_reply():
"""Hitting the hop cap used to surface as "unknown server response", which
sent everyone looking at the payload shape instead of at a model that kept
calling tools and never answered."""
command = {"type": "command", "command": "echo hi", "token": "t"}
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response(command)
with pytest.raises(server_client.ServerError, match="hop cap"):
server_client.converse("hi", on_command=lambda cmd: "ok")
# ── relayed shell commands ──────────────────────────────────────────────────
@pytest.fixture(autouse=True)
def _no_sudo_prompt(monkeypatch):
monkeypatch.setattr(server_client.config, "SUDO_ASKPASS_PROMPT", False)
@pytest.mark.skipif(os.name != "posix", reason="POSIX shell and process groups")
def test_a_successful_command_returns_its_output_and_exit_code():
output = server_client.run_local_command("echo hello; exit 3")
assert output.startswith("[exit 3]")
assert "hello" in output
@pytest.mark.skipif(os.name != "posix", reason="POSIX shell and process groups")
def test_a_timed_out_command_still_reports_what_it_printed():
"""A bare "timed out" tells the model nothing; the last line of output
usually says exactly what it was stuck waiting for."""
output = server_client.run_local_command("echo working on it; sleep 30", timeout=1)
assert "timed out after 1s" in output
assert "working on it" in output
@pytest.mark.skipif(os.name != "posix", reason="POSIX shell and process groups")
def test_a_timed_out_command_takes_its_children_with_it(tmp_path):
"""subprocess.run() would only kill the `sh`, leaving whatever it spawned
running for the rest of the session with no parent watching."""
marker = tmp_path / "ticks"
server_client.run_local_command(
f"(while true; do echo tick >> {marker}; sleep 0.05; done) & sleep 30",
timeout=1,
)
settled = marker.stat().st_size if marker.exists() else 0
time.sleep(0.4)
grew = (marker.stat().st_size if marker.exists() else 0) - settled
assert grew == 0, "a grandchild survived the timeout and is still writing"
+48 -1
View File
@@ -6,7 +6,7 @@ from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.speech_text import for_display, for_speech
from bolt_pet.speech_text import for_display, for_speech, is_question
def test_bold_markers_are_not_spoken():
@@ -57,3 +57,50 @@ def test_blank_and_symbol_only_input():
def test_display_keeps_emoji_but_drops_markdown():
assert for_display("**Done** ✅") == "Done ✅"
assert for_display("* one\n* two") == "• one • two"
def test_is_question_fires_when_a_question_mark_appears_anywhere():
assert is_question("Ready to run a command or start a project?")
assert is_question("It's 7:15 AM. Want me to set a timer?")
assert is_question("What time is it? It's 7:15 AM.")
assert is_question("Can you help me with this? I need a quick answer.")
assert not is_question("It's 7:15 AM on July 23, 2026.")
def test_is_question_ignores_trailing_decoration():
assert is_question("Ready to go? 🚀")
assert is_question('Shall I continue?"')
assert is_question("Want me to fix it? **")
def test_is_question_ignores_question_marks_that_are_not_spoken():
# The '?' here is inside a URL query string, which for_speech strips.
assert not is_question("Docs are at https://example.com/x?y=1")
assert not is_question("")
assert not is_question(None)
def test_abbreviations_are_worded_instead_of_spelled_out():
"""The periods make these look like sentence boundaries, so the voice reads
them letter by letter ("eee gee")."""
assert for_speech("Use a flag, e.g. --force") == "Use a flag, for example force"
assert for_speech("i.e. the config file") == "that is the config file"
assert for_speech("logs, configs, etc.") == "logs, configs, and so on"
assert for_speech("docker vs. podman") == "docker versus podman"
assert for_speech("Fixed in PR #42") == "Fixed in PR number 42"
def test_abbreviation_wording_is_word_bounded():
""""vs" inside a word or filename isn't an abbreviation."""
assert "versus" not in for_speech("the vscode window")
assert "versus" not in for_speech("revs per minute")
# A markdown heading has no digit after the hashes, so it's still a heading.
assert for_speech("## Results") == "Results"
def test_long_option_dashes_are_dropped_but_hyphens_survive():
assert for_speech("run it with --force") == "run it with force"
assert for_speech("check bolt-pet is up-to-date") == "check bolt-pet is up-to-date"
# The rule line is gone; the full stops are _bullets_to_sentences giving the
# voice a pause where the eye saw a line break.
assert for_speech("one\n---\ntwo") == "one. two."
+122
View File
@@ -0,0 +1,122 @@
"""Graphical sudo prompts: command rewriting and helper resolution.
Pure logic no display, no sudo, no password."""
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import sudo_askpass
# ── rewriting ────────────────────────────────────────────────────────────────
@pytest.mark.parametrize(
"command,expected",
[
("sudo apt update", "sudo -A apt update"),
("sudo apt update", "sudo -A apt update"), # spacing preserved
("apt update && sudo apt upgrade", "apt update && sudo -A apt upgrade"),
("ls; sudo reboot", "ls; sudo -A reboot"),
("echo hi | sudo tee /etc/motd", "echo hi | sudo -A tee /etc/motd"),
("sudo systemctl restart x\nsudo systemctl status x",
"sudo -A systemctl restart x\nsudo -A systemctl status x"),
],
)
def test_bare_sudo_gets_the_askpass_flag(command, expected):
assert sudo_askpass.add_askpass_flag(command) == expected
@pytest.mark.parametrize(
"command",
[
"sudo -n apt update", # explicitly non-interactive
"sudo -A apt update", # already asking
"sudo -u bob whoami", # the caller was explicit
"ls -la", # no sudo at all
"echo 'run sudo later'", # inside a quoted string
"pseudo --version", # not the word sudo
],
)
def test_commands_that_must_not_be_rewritten(command):
assert sudo_askpass.add_askpass_flag(command) == command
def test_blank_input():
assert sudo_askpass.add_askpass_flag("") == ""
assert sudo_askpass.add_askpass_flag(None) == ""
@pytest.mark.parametrize(
"command,expected",
[
("sudo apt update", True),
("ls && sudo reboot", True),
("ls -la", False),
("echo 'sudo'", False),
("", False),
],
)
def test_which_commands_get_the_longer_timeout(command, expected):
assert sudo_askpass.needs_password_prompt(command) is expected
# ── helper resolution ────────────────────────────────────────────────────────
def test_a_configured_helper_wins():
found = sudo_askpass.find_helper(
configured="/opt/my-askpass", is_executable=lambda p: p == "/opt/my-askpass",
)
assert found == "/opt/my-askpass"
def test_a_configured_helper_that_is_not_executable_is_not_silently_replaced():
"""Better to have sudo fail than to quietly prompt with something the
user didn't choose."""
assert sudo_askpass.find_helper(
configured="/opt/typo", is_executable=lambda p: False, which=lambda t: "/usr/bin/zenity",
) is None
def test_a_real_askpass_binary_beats_a_generated_wrapper():
found = sudo_askpass.find_helper(
configured="", is_executable=lambda p: p == "/usr/bin/ksshaskpass",
which=lambda tool: "/usr/bin/zenity",
)
assert found == "/usr/bin/ksshaskpass"
def test_falls_back_to_wrapping_a_dialog_tool(tmp_path):
found = sudo_askpass.find_helper(
configured="", is_executable=lambda p: False,
which=lambda tool: "/usr/bin/zenity" if tool == "zenity" else None,
cache_dir=tmp_path,
)
script = tmp_path / "askpass.sh"
assert found == str(script)
assert "zenity --password" in script.read_text()
assert script.stat().st_mode & 0o777 == 0o700 # nobody else edits the password box
def test_no_helper_available_at_all():
assert sudo_askpass.find_helper(
configured="", is_executable=lambda p: False, which=lambda tool: None,
) is None
def test_the_wrapper_passes_sudos_prompt_through():
"""sudo hands the helper its prompt as $1 — it names the account the
password is for, which is worth showing in the dialog."""
script = sudo_askpass.helper_script("/usr/bin/zenity")
assert script.startswith("#!/bin/sh")
assert '"$1"' in script
def test_environment_points_sudo_at_the_helper():
env = sudo_askpass.environment("/tmp/askpass.sh", base={"PATH": "/usr/bin"})
assert env["SUDO_ASKPASS"] == "/tmp/askpass.sh"
assert env["PATH"] == "/usr/bin" # the rest of the environment survives
+26 -1
View File
@@ -8,7 +8,8 @@ import numpy as np
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.audio.tts import chunks_to_int16
from bolt_pet import config as tts_config
from bolt_pet.audio.tts import chunks_to_int16, model_for, voice_for
from bolt_pet.audio.wake_word import NearMissLog
@@ -80,3 +81,27 @@ def test_clear_resets_peak_and_entries():
log.observe(0.45, threshold=0.5, timestamp=1.0)
log.clear()
assert log.entries() == [] and log.peak == 0.0
# ── voice / model selection (server speak_as) ───────────────────────────────
def test_the_override_voice_wins_over_the_configured_one(monkeypatch):
monkeypatch.setattr(tts_config, "ELEVENLABS_VOICE_ID", "DEFAULT")
assert voice_for("VOICE1") == "VOICE1"
assert voice_for("") == "DEFAULT"
assert voice_for(None) == "DEFAULT"
def test_english_replies_in_the_default_voice_use_the_default_model(monkeypatch):
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
assert model_for("all good here", None) == "eleven_flash_v2"
def test_a_picked_voice_or_non_english_text_uses_the_multilingual_model(monkeypatch):
# eleven_flash_v2 is English-only: it would read either of these as
# mangled phonetic English rather than failing outright.
monkeypatch.setattr(tts_config, "ELEVENLABS_MODEL_ID", "eleven_flash_v2")
monkeypatch.setattr(tts_config, "ELEVENLABS_MULTILINGUAL_MODEL_ID", "eleven_flash_v2_5")
assert model_for("all good here", "VOICE1") == "eleven_flash_v2_5"
assert model_for("こんにちは", None) == "eleven_flash_v2_5"
+237
View File
@@ -0,0 +1,237 @@
"""Auto-updater: version comparison, release parsing, and the apply/rollback
dance driven by a fake git (no network, no real repo, nothing checked out)."""
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import updater
# ── version comparison ───────────────────────────────────────────────────────
@pytest.mark.parametrize(
"tag,expected",
[
("v1.2.3", (1, 2, 3)),
("1.2.3", (1, 2, 3)),
("V0.1.0", (0, 1, 0)),
("1.2", (1, 2)),
("1.2.3-beta1", (1, 2, 3)), # suffix ends the parse
("", ()),
("nightly", ()),
],
)
def test_parse_version(tag, expected):
assert updater.parse_version(tag) == expected
@pytest.mark.parametrize(
"candidate,current",
[("0.2.0", "0.1.0"), ("1.0.0", "0.9.9"), ("0.1.1", "0.1"), ("v2.0", "1.9.9")],
)
def test_is_newer_accepts_newer_versions(candidate, current):
assert updater.is_newer(candidate, current)
@pytest.mark.parametrize(
"candidate,current",
[
("0.1.0", "0.1.0"),
("0.1.0", "0.2.0"),
("0.1", "0.1.0"), # zero-padded: equal, not newer
("", "0.1.0"),
("nightly", "0.1.0"), # unparseable is never newer
],
)
def test_is_newer_rejects_same_or_older(candidate, current):
assert not updater.is_newer(candidate, current)
def test_release_parsing_skips_drafts_and_untagged():
assert updater.release_from_payload({"tag_name": "v1.0.0", "draft": True}) is None
assert updater.release_from_payload({"name": "no tag"}) is None
release = updater.release_from_payload({"tag_name": "v1.0.0", "name": "One", "prerelease": True})
assert (release.tag, release.name, release.prerelease) == ("v1.0.0", "One", True)
# ── fake git ─────────────────────────────────────────────────────────────────
class FakeGit:
"""Records every git invocation and answers from a canned script.
*failures* maps a leading-args tuple to the (code, output) it should
return, so a test can make exactly one command fail."""
def __init__(self, ref="main", dirty=False, failures=None, requirements_changed=False,
head="abc1234", tag_commit="def5678"):
self.calls = []
self._ref = ref
self._dirty = dirty
self._failures = failures or {}
self._requirements_changed = requirements_changed
self._head = head
self._tag_commit = tag_commit # equal to head = "already on this tag"
def __call__(self, args):
self.calls.append(list(args))
for prefix, result in self._failures.items():
if tuple(args[: len(prefix)]) == prefix:
return result
head = args[0]
if head == "rev-parse" and args[1] == "--git-dir":
return 0, ".git"
if head == "status":
return 0, " M bolt_pet/config.py" if self._dirty else ""
if head == "symbolic-ref":
return (0, self._ref) if self._ref else (1, "")
if head == "rev-parse":
return 0, self._tag_commit if args[1].startswith("tags/") else self._head
if head == "diff":
return 0, "requirements.txt" if self._requirements_changed else ""
return 0, ""
def commands(self):
"""Just the verbs, for asserting on the sequence."""
return [call[0] for call in self.calls]
def test_apply_update_checks_out_the_tag(tmp_path):
git = FakeGit()
previous = updater.apply_update(
"v1.0.0", run=git, repo=tmp_path, install_deps=False, verify=lambda repo: None
)
assert previous == "main"
assert ["fetch", "--tags", "--prune", "origin"] in git.calls
assert ["checkout", "--force", "tags/v1.0.0"] in git.calls
def test_a_dirty_working_tree_is_left_completely_alone(tmp_path):
git = FakeGit(dirty=True)
with pytest.raises(updater.UpdateError, match="local changes"):
updater.apply_update("v1.0.0", run=git, repo=tmp_path, install_deps=False)
assert "fetch" not in git.commands()
assert "checkout" not in git.commands()
def test_a_non_git_install_refuses_before_touching_anything(tmp_path):
git = FakeGit(failures={("rev-parse", "--git-dir"): (128, "not a repository")})
with pytest.raises(updater.UpdateError, match="not a git clone"):
updater.apply_update("v1.0.0", run=git, repo=tmp_path, install_deps=False)
assert "checkout" not in git.commands()
def test_a_failed_smoke_test_rolls_back_to_the_previous_ref(tmp_path):
git = FakeGit(ref="main")
def broken(repo):
raise updater.UpdateError("the new version failed to import: boom")
with pytest.raises(updater.UpdateError, match="rolled back to main"):
updater.apply_update("v1.0.0", run=git, repo=tmp_path, install_deps=False, verify=broken)
checkouts = [call for call in git.calls if call[0] == "checkout"]
assert checkouts == [["checkout", "--force", "tags/v1.0.0"], ["checkout", "--force", "main"]]
def test_rollback_targets_the_commit_when_head_is_detached(tmp_path):
# No branch to go back to (symbolic-ref fails) — the SHA is the ref.
git = FakeGit(ref="")
with pytest.raises(updater.UpdateError):
updater.apply_update(
"v1.0.0", run=git, repo=tmp_path, install_deps=False,
verify=lambda repo: (_ for _ in ()).throw(RuntimeError("nope")),
)
assert ["checkout", "--force", "abc1234"] in git.calls
def test_a_verify_that_raises_something_unexpected_still_rolls_back(tmp_path):
git = FakeGit()
def exploding(repo):
raise ValueError("not even an UpdateError")
with pytest.raises(updater.UpdateError, match="rolled back"):
updater.apply_update(
"v1.0.0", run=git, repo=tmp_path, install_deps=False, verify=exploding
)
assert ["checkout", "--force", "main"] in git.calls
def test_a_release_tagged_without_bumping_the_version_does_not_loop(tmp_path):
"""Cut a release but forget to bump __version__ in the tagged commit and
every check would see the same "newer" tag: check out (a no-op), restart,
read the old version, repeat a restart loop every check interval."""
git = FakeGit(head="same1234", tag_commit="same1234")
with pytest.raises(updater.UpdateError, match="bump it in the tagged commit"):
updater.apply_update("v0.2.1", run=git, repo=tmp_path, install_deps=False,
verify=lambda repo: None)
assert "checkout" not in git.commands() # nothing moved, so nothing to restart into
def test_an_unknown_tag_is_not_mistaken_for_being_already_on_it(tmp_path):
git = FakeGit(failures={("rev-parse", "tags/"): (128, "unknown revision")})
# The failure key matches by prefix, so make it explicit that a tag we
# can't resolve means "not there yet" rather than "already applied".
assert updater.already_at_tag(git, "v9.9.9") is False
def test_a_failed_fetch_never_moves_the_checkout(tmp_path):
git = FakeGit(failures={("fetch",): (1, "could not resolve host")})
with pytest.raises(updater.UpdateError, match="git fetch failed"):
updater.apply_update("v1.0.0", run=git, repo=tmp_path, install_deps=False)
assert "checkout" not in git.commands()
def test_a_stranded_checkout_is_logged_loudly(tmp_path):
"""Rollback itself failing is the one case a human has to fix by hand."""
git = FakeGit(failures={("checkout", "--force", "main"): (1, "index locked")})
logs = []
with pytest.raises(updater.UpdateError):
updater.apply_update(
"v1.0.0", run=git, repo=tmp_path, install_deps=False, on_log=logs.append,
verify=lambda repo: (_ for _ in ()).throw(updater.UpdateError("bad build")),
)
assert any("ROLLBACK FAILED" in line and "git checkout main" in line for line in logs)
def test_deps_are_only_reinstalled_when_requirements_actually_changed(tmp_path, monkeypatch):
installs = []
monkeypatch.setattr(updater, "_install_deps", lambda repo: installs.append(repo))
updater.apply_update(
"v1.0.0", run=FakeGit(requirements_changed=False), repo=tmp_path,
install_deps=True, verify=lambda repo: None,
)
assert installs == []
updater.apply_update(
"v1.0.0", run=FakeGit(requirements_changed=True), repo=tmp_path,
install_deps=True, verify=lambda repo: None,
)
assert installs == [tmp_path]
# ── release checking ─────────────────────────────────────────────────────────
def test_check_for_update_returns_nothing_when_current(monkeypatch):
monkeypatch.setattr(
updater, "fetch_latest_release",
lambda **kwargs: updater.Release("v0.1.0", "0.1.0", "", False),
)
assert updater.check_for_update(current_version="0.1.0") is None
assert updater.check_for_update(current_version="0.0.9").tag == "v0.1.0"
def test_no_releases_yet_is_not_an_error(monkeypatch):
monkeypatch.setattr(updater, "fetch_latest_release", lambda **kwargs: None)
assert updater.check_for_update(current_version="0.1.0") is None