2 Commits

Author SHA1 Message Date
themajesticmagician 3ee67cb4d6 feat: Enhance local command handling and introduce local intents
- Refactor `run_local_command` to manage subprocesses more effectively, ensuring child processes are terminated on timeout.
- Introduce `_terminate` function to handle process group termination and capture output.
- Implement `_command_output` to format command results with a character limit.
- Add local intent recognition in `intents.py` to handle commands like "stop", "go to sleep", and "come here" without server interaction.
- Normalize user input to match local intents while stripping filler words.
- Update tests to cover new local intent functionality and ensure proper command handling.
- Enhance speech processing to handle abbreviations and improve spoken output clarity.
2026-08-05 18:31:02 -06:00
themajesticmagician 8d4751d80f Update CLAUDE.md with release tagging instructions and enhance is_question logic to detect question marks anywhere in the text 2026-08-05 17:53:32 -06:00
60 changed files with 1265 additions and 2840 deletions
+4 -1
View File
@@ -89,7 +89,10 @@
"Bash(dig +short themajesticnetwork.com)",
"Bash(dig +short api.themajesticnetwork.com)",
"Bash(timeout 900 /tmp/claude-1000/-home-maji-Documents-Bolt-Pet/58f4a7c3-6a92-47ed-9e43-bb2216d8b135/scratchpad/tmnvenv/bin/python -m pytest tests/test_site.py -q -p no:cacheprovider)",
"Bash(curl -s -o /dev/null -w 'HTTP %{http_code} bytes=%{size_download}\\\\n' -m 15 -H 'X-Forwarded-For: 1.2.3.4' -H 'X-Real-IP: 1.2.3.4' -A 'Mozilla/5.0 \\(X11; Linux x86_64\\) Firefox/152.0' https://themajesticnetwork.com/?claude-probe-__TRACKED_VAR__)"
"Bash(curl -s -o /dev/null -w 'HTTP %{http_code} bytes=%{size_download}\\\\n' -m 15 -H 'X-Forwarded-For: 1.2.3.4' -H 'X-Real-IP: 1.2.3.4' -A 'Mozilla/5.0 \\(X11; Linux x86_64\\) Firefox/152.0' https://themajesticnetwork.com/?claude-probe-__TRACKED_VAR__)",
"Bash(QT_QPA_PLATFORM=offscreen timeout 300 .venv/bin/pytest tests/ -q)",
"Bash(QT_QPA_PLATFORM=offscreen timeout 120 .venv/bin/pytest tests/test_intents.py -q)",
"Bash(QT_QPA_PLATFORM=offscreen timeout 120 .venv/bin/pytest tests/test_speech_text.py -q)"
]
}
}
+16 -14
View File
@@ -72,8 +72,17 @@ ELEVENLABS_VOICE_ID=
#VAD_MIN_UTTERANCE_SECONDS=0.4
#VAD_GRACE_SECONDS=4 # how long to wait for you to start talking
# ── Local intents (optional) ────────────────────────────────────────────────
# A short, closed list of things the pet answers itself, with no server round
# trip: "stop", "be quiet", "come here", "go away", "go to sleep", "wake up",
# "say that again", "sit"/"stay", "go for a walk", "use your normal voice".
# Matched whole and exact, and never while you're answering a question Bolt
# asked, so a real request ("stop the docker container") still goes to him.
# Set to false to route absolutely everything through the server.
#LOCAL_INTENTS=true
# ── Follow-up listening (optional) ──────────────────────────────────────────
# When a reply ends on a question, the pet keeps listening for your answer
# When a reply asks you something, the pet keeps listening for your answer
# instead of dropping back to idle and making you say the wake word again.
# FOLLOW_UP_MAX_TURNS caps how many question-and-answer rounds can chain
# without you re-triggering it (0 = no cap) — a stop on runaway loops if the
@@ -103,9 +112,6 @@ ELEVENLABS_VOICE_ID=
#PET_CLICK_THROUGH=false
#PET_EDGE_SNAP=true
#PET_SNAP_MARGIN=48
# Start where you last dragged it. A position on a monitor that's no longer
# connected is ignored, so unplugging a screen can't hide the pet off-desktop.
#PET_REMEMBER_POSITION=true
# ── Barge-in (optional) — interrupt the pet mid-sentence ────────────────────
# BARGE_IN_MODE decides what counts as an interruption:
@@ -141,16 +147,6 @@ ELEVENLABS_VOICE_ID=
# ── Streaming TTS (optional) — starts talking on the first chunk ────────────
#TTS_STREAMING=true
# ── Latency: streaming the reply and the transcript ─────────────────────────
# STREAMING_REPLIES speaks each sentence as the server generates it, instead of
# waiting out the whole model call before the first word. STT_STREAMING sends
# mic frames to Deepgram as you talk, so the transcript is ready the moment you
# stop. Both fall back to the old path automatically if anything goes wrong.
# VAD_SILENCE_END_SEC is the other half: it is dead air on every single turn,
# so 0.8-1.0 feels markedly snappier than the 1.2 default.
#STREAMING_REPLIES=true
#STT_STREAMING=true
# ── Screen context (optional) ───────────────────────────────────────────────
# Sends the focused window's title along with what you said, so "what's this
# error?" has a referent. Text only — no screenshots leave the machine.
@@ -194,6 +190,12 @@ ELEVENLABS_VOICE_ID=
#NOTIFICATION_BRIDGE=false
#NOTIFICATION_FILTER=build|deploy|calendar
#NOTIFICATION_MIN_INTERVAL_SECONDS=60
# Notifications queue while the pet is napping (the heartbeat that forwards them
# doesn't run). These two stop an overnight backlog becoming a monologue at 8am:
# the queue drops its oldest past the limit, and anything staler than the age
# limit is discarded rather than read out.
#NOTIFICATION_QUEUE_LIMIT=20
#NOTIFICATION_MAX_AGE_SECONDS=900
# ── Push-to-talk (optional) ─────────────────────────────────────────────────
# Global hotkey; needs pynput and a session that allows global key hooks
+121 -113
View File
@@ -35,10 +35,6 @@ all suppressed while it's napping (quiet hours / fullscreen DND).
./run.sh # macOS/Linux
run.bat # Windows
# Preflight: is this install actually going to work? (config, mic, keys, sprites…)
python -m bolt_pet --doctor # shallow: no network, no mic
python -m bolt_pet --doctor --deep # contacts the server and opens the microphone
# Run tests (no pytest config file — tests self-insert repo root via sys.path).
# QT_QPA_PLATFORM=offscreen avoids a QApplication segfault on headless/no-display hosts.
QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/
@@ -51,6 +47,12 @@ python scripts/generate_bolt_sprites.py # --out /tmp/x to preview fi
python scripts/slice_spritesheet.py path/to/sheet.png assets/sprites/idle --cols 6 --rows 1
```
**Cutting a release:** bump `__version__` in `bolt_pet/__init__.py` in the same
commit you tag, because that string — not the git history — is what every
already-installed pet compares against the newest Gitea tag (`updater.py`). A
tag without the bump means nobody updates; a bump without the tag means the
next tag looks older than what's running.
There is no lint/build step configured beyond pytest. `cp .env.example .env`
and fill in `BOLT_SERVER_URL` / `DESK_API_KEY` (+ `DEEPGRAM_API_KEY`,
`ELEVENLABS_API_KEY`) before running — without server config the controller
@@ -77,12 +79,41 @@ logs a missing-config message and exits its thread instead of starting.
queued desktop notifications. It owns the live wake-word threshold
(`wake_threshold()` is passed to `listen_for_wake_word` as a *callable* so
the tray slider takes effect mid-listen) and the conversation `history`.
**One thread drives all of it, so failure containment is structural.** Every
entry point that can raise runs inside `_guarded(work, label)`, which logs and
forces the machine back to IDLE (the only state it's always safe to resume
from): the conversation turn, the heartbeat tick — which matters most, since
`on_tick` is the one place control returns to us during a listen that blocks
for minutes, and everything it drives touches the network or shells out — and
the post-restart report. `run()` wraps the lot in try/finally because
`finished` is what `ui/app.py` waits on to quit the thread and to run a
pending `os.execv`; an exception escaping `_loop` used to skip it, so the
failure mode of any bug below was "the pet goes deaf with the mic still open
and the tray won't quit" rather than "one turn failed". `_handle_command` has
the same shape for a different reason: it must **always return a string**,
because the server is blocked on `/desk/tool_result` while it runs and an
exception there means the relay never posts and the server sits out its own
timeout on a turn that can't finish — silent on both ends. Handed back as
command output instead, Bolt can read what broke and say so in the same turn.
- **`server_client.py`** — HTTP client for the desk API, dependency-free
beyond `requests` so it's easy to mock in tests. `converse()` loops relaying
server-issued shell commands (`run_local_command`, executed via
`subprocess.run(shell=True)` as the desktop user, 30s default timeout) via
server-issued shell commands (`run_local_command`, executed via a
`shell=True` `Popen` as the desktop user, 30s default timeout) via
`/desk/tool_result` until the server sends a final `reply` (capped at
`_MAX_RELAY_HOPS`). This is the same "full desktop control" trust model as
`_MAX_RELAY_HOPS` — exhausting which is reported as its own error, because
"unknown server response" sent everyone looking at the payload shape when what
happened is a model that kept calling tools and never answered).
`run_local_command` is `Popen` rather than `subprocess.run` for the timeout
path: the command is a shell, and `run()`'s timeout would kill only that
shell, leaving whatever it spawned (a build, a `tail -f`, an ffmpeg) alive for
the rest of the session with no parent watching — so the child gets its own
process group (`start_new_session`, POSIX) and a timeout SIGTERMs the group,
SIGKILLs it two seconds later, then drains the pipes *with its own timeout* so
a grandchild holding stdout can't turn a timeout into a hang. Whatever the
command printed before it hung is returned alongside the timeout notice, since
the last line usually says exactly what it was stuck waiting for. This is the
same "full desktop control" trust model as
the server repo's other desk clients — commands only ever originate from
the user's own voice/click requests in their own session. A final reply is
returned as a `Reply(text, voice_id, voice_name)` rather than a bare string,
@@ -106,7 +137,8 @@ logs a missing-config message and exits its thread instead of starting.
the name isn't always coming from someone as trusted as the owner. Toggle
off entirely with `RECEIVE_FILES=false`.
- **`audio/`** — `mic.py` (energy-based VAD utterance capture, ported from the
server repo's `bolt_desk.py`), `wake_word.py` (openWakeWord `thunderbolt.onnx`
server repo's `bolt_desk.py`, plus `flush()` — see the note below on the pet
hearing itself), `wake_word.py` (openWakeWord `thunderbolt.onnx`
detection + `NearMissLog` for threshold tuning — see below), `stt.py`
(Deepgram), `tts.py` (ElevenLabs, streaming by default — `stream_pcm()` +
`play_stream()` start playback on the first chunk; `chunks_to_int16()`
@@ -211,6 +243,23 @@ logs a missing-config message and exits its thread instead of starting.
notes below); it's a safer path to the same capability. Pure parsing
(`parse`) is separated from the filesystem I/O (`execute`), matching
pet_actions.py's parse/describe split.
- **`relay_json.py`** — the JSON parser both `filectl` and `dialoguectl` use
instead of `json.loads`, because their payload is hand-typed by a model into
a tool marker and fails in a small, repeatable set of ways (stray quote after
a bare literal, trailing comma, single or smart quotes, Python `True`/`False`,
a markdown fence). Strict parsing already cost a live turn: the call was
rejected, the model re-sent the identical line, was rejected again, and then
told the user "I'll check now" without ever calling anything. So `loads()`
tries strict first, then applies **named, individually-narrow repairs** and
accepts one only if the result parses — and on total failure raises
`RelayJsonError` carrying a caret pointed at the offending character, since
a model can act on a pointed-at fragment but not on "Expecting ',' delimiter:
char 74". Two conventions matter for any new relayed-JSON command: repairs
are **never silent**`parse` stashes them on the action as `_repairs` and
`describe` appends `relay_json.repair_note(...)` to the tool result, so the
model is told it sent something broken while it still has the turn — and new
repairs go in the `_REPAIRS` tuple ordered cheapest/safest first. Tested
inside `tests/test_file_ops.py`, not a file of its own.
- **`screen_context.py`** — active-window title (xprop/xdotool, Win32,
osascript) appended to each utterance via `context_for()`, plus
`is_fullscreen_active()` for do-not-disturb. Text only — the desk API takes
@@ -243,21 +292,18 @@ logs a missing-config message and exits its thread instead of starting.
- **`quiet.py`** — quiet-hours spec parsing (`23:00-08:00`, wraps midnight,
comma-separated). Napping suppresses *proactive* noise and wandering only;
wake word / click / push-to-talk still work.
- **`notifications.py`** — Linux/D-Bus notification bridge. Two latency/loss bugs
fixed 2026-08-02, both in `controller._maybe_heartbeat`: draining was wired to the
60s heartbeat interval rather than the ~1.2s wake tick, and a heartbeat that landed
mid-conversation stamped its own clock *before* checking — burning the slot and
waiting another full interval, repeatedly, which is how a notification could go
unspoken for five or ten minutes. Draining now runs on every tick while IDLE, and the
heartbeat clock only advances when the heartbeat actually runs. Separately, the rate
limit used to *drop* notifications inside its window (a second text a minute later was
silently lost); the filter still gates at queue time but the limit is gone — a burst
is **batched into one turn** instead, same single round trip, no lost messages, capped
by `_MAX_PENDING_NOTIFICATIONS`.
- **`notifications.py` internals** — tails
- **`notifications.py`** — Linux/D-Bus notification bridge: tails
`dbus-monitor`, parses Notify calls (pure `iter_notifications()`), filters
and rate-limits them (`NotificationGate`), and the controller forwards
survivors through `converse()`. Off by default — each one is a round trip.
Note where the queue between the two threads lives: notifications arrive on
the watcher thread and are forwarded from the heartbeat, which **doesn't run
while the pet is napping** — so they accumulate overnight. The controller's
queue is therefore a bounded `deque` stamped on arrival, and the drain
discards anything older than `NOTIFICATION_MAX_AGE_SECONDS` rather than
reading a nine-hour-old backlog out at 8am. A drain that stops early (a nap
starting mid-loop, or the server going down) re-queues what it didn't forward
instead of dropping it, which the original swap-and-return did silently.
- **`sudo_askpass.py`** — makes server-relayed `sudo` usable from a process
with no terminal, by pointing sudo's `SUDO_ASKPASS` at a GUI helper and
rewriting bare `sudo` to `sudo -A` (`add_askpass_flag`, a conservative regex
@@ -290,94 +336,47 @@ logs a missing-config message and exits its thread instead of starting.
- **`hotkey.py`** — global push-to-talk via `pynput`; soft-fails with a logged
reason (Wayland, missing package, macOS permissions) since the wake word is
the primary trigger.
- **`audio/stt_stream.py`** — streaming speech-to-text. The one-shot path waits for
the utterance to end, uploads the whole WAV, then waits again; that second wait is
dead time that grows with how long you spoke. Deepgram's live websocket removes it:
`record_utterance(on_frame=...)` hands each captured frame to a
`StreamingTranscriber`, so by the time the VAD decides you stopped the transcript is
essentially already there. Three deliberate limits: `open()` returning **None is an
ordinary outcome** (no websocket-client, no network, no key) because the full audio is
still buffered and `controller._transcribe` just falls back; the **local VAD still
decides when you stopped** rather than Deepgram's endpointing, since barge-in,
follow-up listening and the grace period are all built on it and coupling them to the
network is not a first-pass change; and the socket is **per-utterance**, because
holding one open across an idle pet bills for silence and dies on the first blip.
Off: `STT_STREAMING=false`.
- **Streamed replies** — `server_client.converse_stream()` reads NDJSON from the desk
API's `/desk/converse_stream` and speaks each sentence as it arrives
(`controller._speak_stream_chunk`), so the wait is time-to-first-sentence instead of
the whole model call. `Reply.spoken` marks a reply whose sentences were already said:
the text is still carried, because the follow-up rule needs to see whether it ended on
a question, but it must not be read out again. A stream that fails *before* anything
was spoken falls back to `converse()` invisibly; one that fails after ends the turn
quietly rather than repeating the first half. Off: `STREAMING_REPLIES=false`.
- **`speech.py`** — everything the pet says, and the policy differences between
kinds of saying. There were four near-copies of this in `controller.py` (a
reply, a holding line, a streamed sentence, a dialogue scene), each repeating
the same dance — transition state, show the bubble, maybe record history,
reset barge-in, call TTS, reset barge-in *again*, resume state, decide
whether to keep the mic open — and they had already drifted apart: one forgot
to arm barge-in, another skipped the follow-up rule. Now the dance is
`Speaker.say()` and the differences are data on a frozen `Utterance`
(`record`, `resume`, `hold_talking`, `follow_up`, `interruptible`), built by
the four classmethods `reply`/`holding`/`stream_chunk`/`scene`. Notable
policies: a holding line is **not** recorded (it's filler; the transcript
should keep the answer) and **not** interruptible (cutting off "give me a
sec" strands the tool already running), and it resumes the state it
interrupted rather than dropping to IDLE, because the turn isn't over. A
streamed chunk stays TALKING so the sprite doesn't flicker between sentences.
Collaborators are injected, so all of this is tested without Qt, audio or a
real state machine. Two subtleties that are bugs waiting to happen: the
barge-in detector is read through a **callable**, not held (it's built after
the Speaker — it needs the mic stream — and swapped when the mode changes; two
copies drifting apart is invisible until the wake model starts hearing the
pet), and `last_detail` is captured *before* the post-playback reset, since
reading it after means every interruption reports zeroed counters.
`follow_up_decision()` is the mic-open rule as a pure function.
- **Lip-sync** — the mouth is driven by the audio, not a timer.
`audio/tts.level_of(frame)` reduces a PCM frame to 0..1 loudness on a **sqrt
curve** (speech sits well below peak most of the time, so a linear map leaves
the mouth barely open during normal talking), `envelope()` does the same for a
whole clip, and playback calls the `on_level` hook per chunk. That travels
`Speaker``controller.mouth` (a Signal) → `PetWindow.set_mouth`, and
`_mouth_frame()` indexes the talking frames directly — which works because
`generate_bolt_sprites.py` draws them as an **openness ramp** (closed first,
widest last) rather than an arbitrary loop. Levels going stale
(`_MOUTH_STALE_SECONDS`) hands control back to the ordinary animation, so
offline TTS — which has no envelope — degrades to the timed loop instead of
freezing the mouth mid-syllable.
- **`window_state.py`** — where the pet was left, so it starts there.
`~/.cache/bolt-pet/window.json`, atomic write on drag-end, every function
swallows its own errors (a corrupt state file must mean the default corner,
never a pet that won't start). `is_visible_on()` re-validates against the
*current* screen layout on load, because the common case for a stale position
is exactly the dangerous one: the pet was last on a monitor that is now
unplugged, and restoring it faithfully puts it somewhere unreachable. Takes
plain rectangles rather than importing Qt. Off: `PET_REMEMBER_POSITION=false`.
- **`doctor.py`** — `python -m bolt_pet --doctor`, a preflight, written after a
week of debugging things one command would have shown: a venv whose python
was a zero-byte file, an OCR engine never installed, a wrong proxy header.
Twelve independent checks, each reporting ok / warn (degraded but working) /
fail, and each saying **what to do about it**`screen reading: warn` is
useless alone, `apt install tesseract-ocr` is the whole point. Nothing raises:
a doctor that crashes on a broken install is diagnosing the wrong patient, so
`run()` catches per-check and a failed check becomes a FAIL row rather than a
traceback. Shallow by default (no network, no mic) since it's the first thing
you reach for when the network is what's broken; `--deep` actually contacts
the server and opens the microphone.
- **`speech_text.py`** — sanitizes server replies before they're heard/shown.
`for_speech()` (called inside `tts.speak()`, so every path to the speakers is
covered) strips markdown, emoji, URLs and stray symbols the voice would read
literally ("asterisk asterisk"), turns bullet lists into full sentences, and
words a few symbols (`&` → "and"). `for_display()` is the looser version for
words a few symbols (`&` → "and"), abbreviations the voice would spell out
letter by letter (`e.g.` → "for example", `etc.` → "and so on") and a long
option's leading `--` (heard as "dash dash force"; the single hyphen has to
survive for "bolt-pet"). `for_display()` is the looser version for
the speech bubble — markdown syntax gone, emoji kept. `is_question()` decides
whether a reply leaves the pet waiting on an answer: it tests the *spoken*
form (so a '?' inside a stripped code block or URL doesn't count) and only a
trailing one counts, since a question asked in passing isn't awaiting a
reply. `speech.follow_up_decision` uses it to keep listening without the
form (so a '?' inside a stripped code block or URL doesn't count) and a '?'
**anywhere** counts. That last part was once trailing-only, on the theory that
"What time is it? It's 7:15." isn't awaiting a reply — true of that sentence
and wrong more often, since Bolt routinely asks and then keeps talking ("Want
me to fix it? I'd start with the config"), which is the case that actually
costs you a wake word. The asymmetry is the argument: an unwanted extra listen
ends itself on `VAD_GRACE_SECONDS` of silence, a missed one makes you start
over. `controller._should_follow_up` uses it to keep listening without the
wake word, capped by `FOLLOW_UP_MAX_TURNS` so a server that ends every reply
with a question can't loop forever off mic noise. Pure string logic, no
Qt/audio imports.
- **`intents.py`** — the handful of utterances answered *without* the server.
"stop", "come here", "go to sleep", "say that again", "use your normal voice"
are commands to the body, and routing them through the desk API costs two to
four seconds and three network hops to make the pet walk left — and only works
if the server's prompt happens to advertise the matching `petctl` verb (which
is why `voice reset` needs a block in `ai/desk_api.py`'s pet prompt; see the
Voices section). Recognising the phrase here removes both the latency and that
coupling. The design problem is *not stealing real requests*, and three rules
cover it: whole-utterance exact match after normalisation (so "stop" is an
intent and "stop the docker container" is a question for Bolt), a closed table
with nothing arguable in it, and **never on a follow-up turn** — if Bolt just
asked you something your answer is his, and swallowing "never mind" locally
would leave the server holding a question it never got an answer to. Both
sides of the comparison go through `normalize()` (the table is canonicalised
at import, and `_build()` refuses to build one where two intents claim the
same normalised phrase, or where a phrase reduces to "" and would match pure
filler like "hey bolt"). Actions come back in the **same shape
`pet_actions.parse` produces**, so `PetWindow.apply_action` needs no new
vocabulary; the effects live in `controller._handle_local_intent`. Off switch:
`LOCAL_INTENTS=false`.
- **`ui/`** — `app.py` wires `QApplication` + `PetWindow` + `PetTray` + the
history/tuner windows + the push-to-talk hotkey + the controller thread
together; `pet_window.py` is the frameless/translucent/always-on-top sprite
@@ -484,6 +483,29 @@ the times it nearly heard you), and its slider is read per frame because
`listen_for_wake_word` accepts a callable threshold. Set the threshold just
under the peak you can hit reliably, then persist it in `.env`.
### The mic keeps recording while nothing is reading it
Same family of bug as the openwakeword one above, one layer down: PortAudio
captures into a ring buffer continuously, so audio from a stretch where the
pipeline thread was busy elsewhere is still queued when the next read happens.
It bites in exactly one place. At the end of a reply that asked you something,
`_speak` sets `_talk_now` and the next turn starts recording immediately — with
the tail of the pet's own TTS sitting in that buffer, above the VAD threshold.
The VAD takes it for the start of your answer, Deepgram transcribes it, and Bolt
is handed his own last sentence as if you had said it. With barge-in on the
detector was draining the stream during playback so the window is small; with
`BARGE_IN=false` nothing drains it at all.
`mic.flush(stream)` drops what's buffered, and `_speak` calls it on the
follow-up branch only. **That placement is the whole correctness argument**
flushing is only safe where the buffer is known to hold nothing *you* said:
playback ran to completion, so if you had spoken, barge-in would have cut it and
taken the interrupted branch instead. Never flush before a wake-triggered
recording, where the rest of "thunderbolt, what time is it" is legitimately
queued and dropping it clips the request. A single call is bounded by
`max_seconds` so it can't chase a stream filling as fast as it drains, and it
no-ops on a stream with no `read_available` (i.e. every fake stream in tests).
### Testing conventions
`tests/` covers pure logic only (state machine, wake-word scoring loop, mic
@@ -504,20 +526,6 @@ HTTP chunk reassembly by `chunks_to_int16`, emote motion by
`screen_context.context_for` / `is_fullscreen_active`, otherwise they shell
out to xprop on a headless box.
**`test_pipeline_smoke.py` is the exception, deliberately.** Unit tests inject
a fake at the seam they care about, and a run of production bugs — notifications
sitting unspoken for minutes, the pet saying things twice, `[laughing]` read
aloud, a device command arriving as prose — got through with every one of them
green, because each was an *interaction* between two individually-correct
units. So that file stands up a real threaded HTTP server on a loopback port,
speaks the desk protocol at it, and drives whole turns through the real
`server_client` (including NDJSON streaming), the real controller and the real
state machine, faking only the mic stream and the speakers. When a bug crosses
a module boundary, add the case there; when it lives inside one module, the
pure-function pattern above is still the cheaper test. It is also worth
mutation-checking a new case — break the source line it is meant to catch and
confirm it actually goes red.
## Security notes
The server can relay a shell command back to this machine to execute as the
+5 -38
View File
@@ -140,33 +140,6 @@ near misses — frames that scored just under the threshold — and the slider
takes effect immediately, mid-listen. Set the threshold just below the peak
you can hit reliably, then write it into `.env` as `WAKE_WORD_THRESHOLD`.
## Something not working?
```bash
python -m bolt_pet --doctor
```
Checks the things that make the pet look broken in ways that don't point at
themselves — missing server config (the controller exits its thread at startup,
so the pet appears alive and simply never answers), no input device, no OCR
engine behind `petctl read`, a wake model that isn't where `.env` says, a
silence timeout long enough to feel like lag. Each line says what to do about
it, not just what's wrong. It doesn't touch the network or open the microphone
unless you add `--deep`, so it's safe to run when the network is the suspect.
## Little things
The pet **remembers where you left it** — drag it somewhere deliberate and
that's where it starts next time. If that position is on a monitor you've since
unplugged it goes back to the default corner rather than restoring itself
somewhere off-screen. `PET_REMEMBER_POSITION=false` to always start in the
corner.
Its **mouth moves with the actual audio** rather than flapping on a timer: the
PCM going to the speakers is reduced to a loudness per frame and that picks the
talking sprite, so the pet shuts up when the voice pauses. Offline `pyttsx3`
playback has no waveform to follow, so it falls back to the timed loop.
## Project layout
```
@@ -175,7 +148,6 @@ bolt_pet/
state.py PetState enum + a small transition-checked state machine
server_client.py /desk/converse, /desk/tool_result, /desk/report_status
controller.py the pipeline: wake word -> STT -> server -> TTS, on a QThread
speech.py what the pet says and how each kind of saying behaves
speech_text.py strips markdown/emoji/URLs so the voice never says "asterisk"
pet_actions.py petctl move/emote/say/wander/nap parsing
screen_context.py active-window title + fullscreen detection
@@ -183,15 +155,11 @@ bolt_pet/
notifications.py desktop notification bridge (Linux/D-Bus)
history.py rolling conversation transcript
hotkey.py global push-to-talk (pynput, optional)
window_state.py remembers where you left the pet
doctor.py `--doctor` preflight: is this install going to work?
audio/
mic.py input stream + energy-based VAD utterance capture
wake_word.py openWakeWord thunderbolt.onnx detection (see above)
stt.py Deepgram (one-shot)
stt_stream.py Deepgram live websocket — transcribes while you speak
tts.py ElevenLabs streaming PCM, offline pyttsx3 fallback,
plus the loudness envelope that drives the mouth
stt.py Deepgram
tts.py ElevenLabs streaming PCM, offline pyttsx3 fallback
barge_in.py "you started talking" detector, to cut playback short
ui/
app.py wires QApplication + window + tray + controller thread together
@@ -204,10 +172,8 @@ bolt_pet/
scripts/
slice_spritesheet.py cuts a grid sprite sheet into the per-frame convention
tests/ pure-logic unit tests (state machine, wake-phrase
matching, HTTP client against mocks) plus one
end-to-end smoke test that drives whole turns against
a real local HTTP server — nothing here needs real
audio hardware or a display
matching, HTTP client against mocks) — nothing here
needs real audio hardware or a display
```
## Security notes
@@ -231,6 +197,7 @@ notifications. No screenshots or images are ever sent.
## Known limitations / not-yet-done
- Pet screen position isn't persisted across restarts.
- Wandering is a straight walk to a random point — no Shimeji-style physics,
wall-climbing or falling.
- Push-to-talk and the notification bridge are platform-limited: the hotkey
+3 -10
View File
@@ -1,15 +1,8 @@
"""Entry point: python -m bolt_pet [--doctor [--deep]]"""
"""Entry point: python -m bolt_pet"""
import sys
from .ui.app import run
if __name__ == "__main__":
if "--doctor" in sys.argv:
# Imported inside the branch, not at module scope: the doctor exists to
# diagnose installs where importing the UI would itself blow up.
from .doctor import main as doctor
sys.exit(doctor(sys.argv[1:]))
from .ui.app import run
sys.exit(run())
Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 57 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 59 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 56 KiB

After

Width:  |  Height:  |  Size: 56 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 65 KiB

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 64 KiB

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 64 KiB

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 63 KiB

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 63 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 59 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 59 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 59 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

After

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 58 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 57 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 61 KiB

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 61 KiB

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 62 KiB

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 62 KiB

After

Width:  |  Height:  |  Size: 62 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 59 KiB

After

Width:  |  Height:  |  Size: 61 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 61 KiB

After

Width:  |  Height:  |  Size: 61 KiB

+31 -17
View File
@@ -37,18 +37,40 @@ def rms(frame: np.ndarray) -> float:
return float(np.sqrt(np.mean(frame.astype(np.float64) ** 2)))
def _emit(on_frame, frame) -> None:
"""Hand a frame to a listener without letting it break the capture.
def flush(stream, max_seconds: float = 10.0, sample_rate: int = config.SAMPLE_RATE) -> int:
"""Throw away whatever is already sitting in the mic's buffer. Returns the
number of frames dropped.
Streaming STT is an optimisation riding along with recording; if its
socket dies mid-utterance the recording must carry on untouched, because
the one-shot fallback is about to need the full buffer."""
if on_frame is None:
return
PortAudio keeps capturing into a ring buffer while nothing is reading it, so
audio recorded during a long blocking stretch is still queued when the next
read happens. That matters exactly once: at the end of a reply the pet is
about to listen for an answer, and the last fraction of a second of its own
TTS is in that buffer. It's above the VAD threshold, so `record_utterance`
treats it as the start of your answer, Deepgram transcribes it, and Bolt is
handed his own sentence as if you had said it. With barge-in on, the
detector was draining the stream during playback and the window is small;
with `BARGE_IN=false` nothing drains it at all.
Only safe where the buffer is known to hold *nothing you said* — never
before a wake-triggered recording, where the rest of "thunderbolt, what
time is it" is legitimately queued and dropping it clips the request.
*max_seconds* bounds a single call so this can't chase a stream that's
filling as fast as it's read. Best-effort: a fake stream in tests has no
`read_available` and this is a no-op, which is the correct behaviour for
one."""
try:
on_frame(frame)
available = int(getattr(stream, "read_available", 0) or 0)
except (TypeError, ValueError):
return 0
if available <= 0:
return 0
frames = min(available, int(max_seconds * sample_rate))
try:
stream.read(frames)
except Exception:
pass
return 0 # a mid-flush device error is the reader's problem, not ours
return frames
def record_utterance(
@@ -61,7 +83,6 @@ def record_utterance(
grace_s: float = None,
frame_len: int = config.FRAME_LEN,
sample_rate: int = config.SAMPLE_RATE,
on_frame=None,
) -> Optional[np.ndarray]:
"""Capture one utterance from *stream*: wait for speech to start, stop
after trailing silence. Returns None if nothing usable was heard.
@@ -70,11 +91,6 @@ def record_utterance(
(e.g. the pet window was closed) without needing threading primitives
baked into this function.
*on_frame*, if given, is called with each captured frame while speech is
in progress — that is how streaming STT transcribes as you talk rather
than after (see audio/stt_stream.py). It is fire-and-forget: this function
still returns the full buffer, so a failed stream costs nothing.
*grace_s* is how long to wait for speech to *begin* before giving up.
The controller stretches it for follow-up questions, where you're being
asked something and need a moment to think rather than having just said
@@ -103,12 +119,10 @@ def record_utterance(
if frame_rms >= rms_threshold:
started = True
frames.append(frame)
_emit(on_frame, frame)
elif waited > grace_frames:
return None # woke it up but said nothing
continue
frames.append(frame)
_emit(on_frame, frame)
if frame_rms < rms_threshold:
silence_frames += 1
if silence_frames >= silence_limit:
-181
View File
@@ -1,181 +0,0 @@
"""Streaming speech-to-text — transcribing *while* you talk, not after.
The one-shot path (`stt.transcribe`) waits for the utterance to finish, then
uploads the whole WAV and waits again. That second wait is dead time between
you stopping and the pet reacting, and it grows with the length of what you
said — a thirty-second question costs noticeably more than a five-second one.
Deepgram's live endpoint removes it: frames go up as they are captured, so by
the time the VAD decides you have stopped, the transcript is essentially
already there. Same model, same account, same accuracy — the difference is
purely when the work happens.
Design constraints that shaped this:
- **Failure must be invisible.** No websocket, no network, a mid-utterance
disconnect — all of it falls back to the one-shot path, which still has the
full audio buffered. Streaming is an optimisation, never a dependency, so
`open()` returning None is an ordinary outcome rather than an error.
- **The VAD still decides when you stopped.** Deepgram has its own endpointing
and using it would save more, but it would also move a decision the rest of
the pipeline is built around (barge-in, follow-up listening, the grace
period) into a remote service. Not worth coupling those to the network on
the first pass.
- **The socket is per-utterance.** Holding one open across an idle pet would
bill for silence and drop on the first network blip; opening one takes
~100ms, which is already inside the time it takes a person to start talking.
The websocket client is injectable, so the whole protocol — send frames, read
`is_final` transcripts, close, take the result — is tested without a network.
"""
from __future__ import annotations
import json
import logging
import threading
from typing import Callable, Optional
import numpy as np
from .. import config
logger = logging.getLogger("bolt_pet.stt_stream")
_ENDPOINT = (
"wss://api.deepgram.com/v1/listen"
"?encoding=linear16&channels=1&sample_rate={rate}&model={model}"
"&language=en&smart_format=true&interim_results=false"
)
def available() -> bool:
"""Whether streaming STT can even be attempted in this install."""
if not config.STT_STREAMING or not config.DEEPGRAM_API_KEY:
return False
try:
import websocket # noqa: F401 (websocket-client)
return True
except Exception:
return False
class StreamingTranscriber:
"""One utterance's worth of live transcription.
Usage mirrors how the capture loop already works — feed frames as they
arrive, then ask what was said:
session = StreamingTranscriber.open()
...
session.feed(frame) # per mic frame, non-blocking
text = session.finish() # after the VAD says you stopped
"""
def __init__(self, socket, *, sample_rate: int = None):
self._socket = socket
self._sample_rate = sample_rate or config.SAMPLE_RATE
self._transcript: list[str] = []
self._lock = threading.Lock()
self._closed = False
self._reader = threading.Thread(
target=self._read_loop, name="stt-stream-reader", daemon=True)
self._reader.start()
# -- lifecycle ----------------------------------------------------------
@classmethod
def open(cls, *, connect: Optional[Callable] = None,
sample_rate: int = None) -> Optional["StreamingTranscriber"]:
"""Connect, or return None if streaming isn't possible right now.
None is a normal outcome, not a failure: the caller keeps the audio and
falls back to the one-shot upload."""
if connect is None and not available():
return None
rate = sample_rate or config.SAMPLE_RATE
try:
if connect is not None:
socket = connect()
else:
import websocket
socket = websocket.create_connection(
_ENDPOINT.format(rate=rate, model=config.DEEPGRAM_MODEL),
header={"Authorization": f"Token {config.DEEPGRAM_API_KEY}"},
timeout=10,
)
return cls(socket, sample_rate=rate)
except Exception as exc:
logger.info("Streaming STT unavailable (%s) — using the one-shot path.", exc)
return None
def feed(self, frame: np.ndarray) -> None:
"""Send one captured frame. Never raises — a dead socket just means the
fallback will do the work."""
if self._closed:
return
try:
self._socket.send_binary(np.asarray(frame, dtype=np.int16).tobytes())
except Exception:
logger.debug("Streaming STT send failed; abandoning the stream", exc_info=True)
self._closed = True
def finish(self, timeout: float = 3.0) -> str:
"""Close the stream and return whatever was transcribed.
Deepgram flushes its final results after the close frame, so this waits
briefly for the reader — bounded, because a hung socket must not hold
up the reply."""
if not self._closed:
try:
self._socket.send(json.dumps({"type": "CloseStream"}))
except Exception:
pass
self._closed = True
self._reader.join(timeout=timeout)
try:
self._socket.close()
except Exception:
pass
with self._lock:
return " ".join(part for part in self._transcript if part).strip()
# -- the reader ---------------------------------------------------------
def _read_loop(self) -> None:
while not self._closed:
try:
message = self._socket.recv()
except Exception:
break
if not message:
break
text, is_final = self._parse(message)
if text and is_final:
with self._lock:
self._transcript.append(text)
@staticmethod
def _parse(message) -> tuple[str, bool]:
"""Pull (text, is_final) out of a Deepgram results frame.
Tolerant on purpose: anything unrecognised is ignored rather than
raising on the reader thread, where an exception would silently kill
transcription for the rest of the utterance."""
try:
if isinstance(message, bytes):
message = message.decode("utf-8", "ignore")
data = json.loads(message)
except (TypeError, ValueError):
return "", False
if not isinstance(data, dict):
return "", False
alternatives = (
((data.get("channel") or {}).get("alternatives") or [])
if data.get("type") in (None, "Results") else []
)
if not alternatives:
return "", False
text = str((alternatives[0] or {}).get("transcript") or "").strip()
return text, bool(data.get("is_final") or data.get("speech_final"))
+7 -64
View File
@@ -14,7 +14,6 @@ fallback has no such concept and always sounds like itself.
from __future__ import annotations
import time
from typing import Iterable, Iterator, Optional
import numpy as np
@@ -163,81 +162,32 @@ def synthesize_dialogue(
return pcm, config.TTS_SAMPLE_RATE
# ── how loud is it right now ────────────────────────────────────────────────
# The PCM is already decoded here on its way to the speakers, so the amplitude
# envelope is free — and it is exactly what a mouth needs to move in time with
# speech. Throwing it away and animating the mouth on a timer instead is why
# most talking sprites look dubbed.
# int16 RMS that counts as "mouth fully open". Speech peaks around 8-12k;
# 6000 keeps normal talking in the upper half of the range without clipping
# every syllable to wide-open.
_LOUD_RMS = 6000.0
def level_of(frame: np.ndarray) -> float:
"""0..1 loudness for one chunk of PCM.
Square-rooted because perceived loudness is not linear in amplitude a
linear mapping leaves the mouth barely moving through ordinary speech."""
if frame is None or len(frame) == 0:
return 0.0
rms = float(np.sqrt(np.mean(np.square(frame.astype(np.float32)))))
return float(min(1.0, (rms / _LOUD_RMS) ** 0.5))
def envelope(pcm: np.ndarray, sample_rate: int, fps: int = 30) -> list:
"""Per-frame loudness for a whole clip, for playback that isn't streamed."""
if pcm is None or len(pcm) == 0:
return []
window = max(1, int(sample_rate / max(1, fps)))
return [level_of(pcm[start:start + window]) for start in range(0, len(pcm), window)]
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None,
on_level=None) -> bool:
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None) -> bool:
"""Play a whole clip. Returns True if it finished, False if *should_stop*
(barge-in) cut it short. *should_stop* is polled while audio plays each
poll consumes one mic frame, which is what paces this loop."""
import sounddevice as sd
levels = envelope(pcm, sample_rate) if on_level is not None else []
started = time.monotonic()
sd.play(pcm, samplerate=sample_rate, device=config.SPEAKER_DEVICE)
if not blocking:
return True
if should_stop is None and on_level is None:
if should_stop is None:
sd.wait()
return True
while True:
if levels:
# Indexed by elapsed time rather than by chunk, because this path
# hands the whole clip to the device at once and never sees it
# again — wall clock is the only position we have.
index = int((time.monotonic() - started) * 30)
if index < len(levels):
try:
on_level(levels[index])
except Exception:
levels = []
try:
if not sd.get_stream().active:
break
except Exception:
break # stream already torn down — playback is over
if should_stop is not None and should_stop():
if should_stop():
sd.stop()
return False
return True
def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None,
on_level=None) -> bool:
"""Play int16 chunks as they arrive. Returns False if interrupted.
*on_level* receives each chunk's loudness (0..1) just before it is written,
which is what drives the mouth: the sprite is animated by the same audio
the speakers are getting, not by a guess about how long a word takes."""
def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None) -> bool:
"""Play int16 chunks as they arrive. Returns False if interrupted."""
import sounddevice as sd
with sd.OutputStream(
@@ -249,11 +199,6 @@ def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None
# now, not at the end of the buffered chunk.
out.abort()
return False
if on_level is not None:
try:
on_level(level_of(chunk))
except Exception:
on_level = None # a broken listener must not stop playback
out.write(chunk)
return True
@@ -268,8 +213,7 @@ def speak_offline(text: str) -> None:
engine.runAndWait()
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None,
on_level=None) -> bool:
def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] = None) -> bool:
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
exception when ElevenLabs fails (useful for logging) a fallback still
@@ -289,14 +233,13 @@ def speak(text: str, on_error=None, should_stop=None, voice_id: Optional[str] =
stream_pcm(text, voice_id=voice_id),
config.TTS_SAMPLE_RATE,
should_stop=should_stop,
on_level=on_level,
)
except TtsError as exc:
if on_error is not None:
on_error(exc)
try:
pcm, sample_rate = synthesize_pcm(text, voice_id=voice_id)
return play_pcm(pcm, sample_rate, should_stop=should_stop, on_level=on_level)
return play_pcm(pcm, sample_rate, should_stop=should_stop)
except TtsError as exc:
if on_error is not None:
on_error(exc)
+16 -16
View File
@@ -58,11 +58,6 @@ WAKE_CHECK_INTERVAL_SECONDS = float(os.environ.get("WAKE_CHECK_INTERVAL_SECONDS"
DEEPGRAM_API_KEY = os.environ.get("DEEPGRAM_API_KEY", "")
DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3")
# Transcribe *while* you talk instead of uploading the finished clip: frames go
# up as they are captured, so the transcript is ready the moment the VAD says
# you stopped. Needs `websocket-client`; falls back to the one-shot upload
# whenever it can't connect, so turning it on can only help.
STT_STREAMING = os.environ.get("STT_STREAMING", "true").lower() in ("1", "true", "yes", "on")
# ── TTS (ElevenLabs, requested as raw PCM so playback needs no external
# player binary — cross-platform via sounddevice instead of shelling out to
@@ -128,6 +123,14 @@ FOLLOW_UP_LISTEN = os.environ.get("FOLLOW_UP_LISTEN", "true").lower() in ("1", "
FOLLOW_UP_MAX_TURNS = int(os.environ.get("FOLLOW_UP_MAX_TURNS", "10"))
FOLLOW_UP_GRACE_SECONDS = float(os.environ.get("FOLLOW_UP_GRACE_SECONDS", "7"))
# ── local intents ───────────────────────────────────────────────────────────
# A short, closed list of utterances the pet answers itself instead of paying a
# server round trip for: "stop", "come here", "go to sleep", "say that again",
# "use your normal voice". Matched whole and exact (see intents.py), never
# during a follow-up turn, so a real request is never swallowed. Turn it off to
# route absolutely everything through Bolt.
LOCAL_INTENTS = os.environ.get("LOCAL_INTENTS", "true").lower() in ("1", "true", "yes", "on")
COMMAND_TIMEOUT_SECONDS = int(os.environ.get("COMMAND_TIMEOUT_SECONDS", "30"))
# ── sudo password prompts ───────────────────────────────────────────────────
@@ -173,13 +176,6 @@ BARGE_IN_FRAMES = int(os.environ.get("BARGE_IN_FRAMES", "4")) # consecutive lou
# waking the pet from idle (useful if Bolt's own voice trips the model).
BARGE_IN_WAKE_THRESHOLD = float(os.environ.get("BARGE_IN_WAKE_THRESHOLD") or 0) or None
# ── streaming the reply ─────────────────────────────────────────────────────
# Speak each sentence as the server produces it, instead of waiting out the
# whole model call before the first word. Falls back to the ordinary
# request/response path automatically if the server has no streaming endpoint
# or the stream fails before anything has been spoken.
STREAMING_REPLIES = os.environ.get("STREAMING_REPLIES", "true").lower() in ("1", "true", "yes", "on")
# ── streaming TTS ───────────────────────────────────────────────────────────
# ElevenLabs' /stream endpoint + chunked playback: the pet starts talking
# after the first PCM chunk instead of after the whole clip is synthesized.
@@ -233,6 +229,14 @@ NOTIFICATION_BRIDGE = os.environ.get("NOTIFICATION_BRIDGE", "false").lower() in
# Regex matched against "<app>: <summary> <body>"; empty means "everything".
NOTIFICATION_FILTER = os.environ.get("NOTIFICATION_FILTER", "")
NOTIFICATION_MIN_INTERVAL_SECONDS = float(os.environ.get("NOTIFICATION_MIN_INTERVAL_SECONDS", "60"))
# Notifications arrive on the watcher thread and are forwarded from the
# heartbeat, which doesn't run while the pet is napping — so they queue. Both
# limits exist to stop an overnight backlog turning into a burst of round trips
# and a monologue at 8am: the queue is bounded (oldest dropped first) and
# anything staler than the age limit is discarded at drain time, because
# "Firefox finished downloading" is not news nine hours later.
NOTIFICATION_QUEUE_LIMIT = int(os.environ.get("NOTIFICATION_QUEUE_LIMIT", "20"))
NOTIFICATION_MAX_AGE_SECONDS = float(os.environ.get("NOTIFICATION_MAX_AGE_SECONDS", "900"))
# ── file delivery ────────────────────────────────────────────────────────
# The server's deliver_files tool (ai/desk_api.py in the main tmn-api repo)
@@ -319,10 +323,6 @@ PET_WANDER_MARGIN = int(os.environ.get("PET_WANDER_MARGIN", "20")) # keep off s
PET_SHAPED_INPUT = os.environ.get("PET_SHAPED_INPUT", "true").lower() in ("1", "true", "yes", "on")
PET_CLICK_THROUGH = os.environ.get("PET_CLICK_THROUGH", "false").lower() in ("1", "true", "yes", "on")
# Snap flush to a screen edge when dropped/parked within this many pixels of it.
# Start where it was left rather than in the bottom-right corner. Ignored when
# PET_START_X/Y pin it explicitly, and a saved position on a monitor that is no
# longer plugged in is discarded rather than hiding the pet offscreen.
PET_REMEMBER_POSITION = os.environ.get("PET_REMEMBER_POSITION", "true").lower() in ("1", "true", "yes", "on")
PET_EDGE_SNAP = os.environ.get("PET_EDGE_SNAP", "true").lower() in ("1", "true", "yes", "on")
PET_SNAP_MARGIN = int(os.environ.get("PET_SNAP_MARGIN", "48"))
+241 -166
View File
@@ -15,25 +15,21 @@ from __future__ import annotations
import threading
import time
from collections import deque
from typing import Optional
from PySide6.QtCore import QObject, Signal
from . import (
config, dialogue as dialogue_mod, speech, file_delivery, file_ops,
history as history_mod, monitors as monitors_mod, notifications,
pet_actions, quiet, screen_context, screen_text, self_restart,
server_client, speech_text, updater,
config, dialogue as dialogue_mod, file_delivery, file_ops,
history as history_mod, intents as intents_mod, monitors as monitors_mod,
notifications, pet_actions, quiet, screen_context, screen_text,
self_restart, server_client, speech_text, updater,
)
from . import __version__
from .audio import barge_in, mic, stt, stt_stream, tts, wake_word
from .audio import barge_in, mic, stt, tts, wake_word
from .state import PetState, PetStateMachine
# A burst while the pet is busy is batched into one turn, but the queue still
# needs a ceiling — a notification storm must not become an unbounded backlog
# that gets read out minutes later.
_MAX_PENDING_NOTIFICATIONS = 12
# How often to re-check whether the pet should be napping. The fullscreen
# probe shells out to xprop, so this deliberately isn't every heartbeat tick.
_NAP_CHECK_INTERVAL_SECONDS = 10.0
@@ -46,7 +42,6 @@ class PetController(QObject):
action = Signal(dict) # parsed petctl action for the UI to perform
napping = Signal(bool) # quiet hours / fullscreen do-not-disturb
voice_changed = Signal(str) # name of the server-picked voice ("" = default)
mouth = Signal(float) # 0..1 speech loudness, for lip-sync while talking
restart_requested = Signal(str) # version we just updated to
finished = Signal()
@@ -83,17 +78,6 @@ class PetController(QObject):
self._voice_name = ""
self._barge_in: Optional[barge_in.BargeInDetector] = None
# Everything the pet says goes through here; see speech.py for why the
# four hand-rolled copies of this became one.
self._speaker = speech.Speaker(
state=self._state, tts=tts,
history=lambda text: self.history.add(history_mod.PET, text, time.time()),
on_said=self.said.emit, on_log=self.log.emit,
barge_in=lambda: self._barge_in,
detail_of=self._barge_in_detail,
voice_id=lambda: self._voice_id,
on_level=self.mouth.emit,
)
self._napping = False
self._nap_forced: Optional[bool] = None # petctl nap on/off overrides the schedule
self._last_nap_check = 0.0
@@ -115,7 +99,14 @@ class PetController(QObject):
self._notification_gate = notifications.NotificationGate(
config.NOTIFICATION_FILTER, config.NOTIFICATION_MIN_INTERVAL_SECONDS
)
self._pending_notifications: list[notifications.Notification] = []
# Bounded, and stamped on arrival: the drain only runs from the
# heartbeat, which doesn't run while napping, so this fills up
# overnight. maxlen drops the oldest rather than growing without limit,
# and the stamp lets the drain discard a backlog nobody wants read out
# at 8am (see _drain_notifications).
self._pending_notifications: deque[tuple[float, notifications.Notification]] = deque(
maxlen=max(1, config.NOTIFICATION_QUEUE_LIMIT)
)
self._notification_lock = threading.Lock()
# ── external controls (safe to call from the Qt/UI thread) ─────────
@@ -204,6 +195,13 @@ class PetController(QObject):
)
self.log.emit(f"Barge-in: {config.BARGE_IN_MODE} mode.")
# Everything past here is in try/finally because `finished` is what
# ui/app.py waits on to quit the QThread and to run a pending
# os.execv. An exception escaping _loop used to skip it, leaving the
# thread wedged with the mic still open and no restart — so the failure
# mode of any bug below was "the pet goes deaf and the tray won't quit"
# rather than "one turn failed".
try:
with self._stream:
try:
health = server_client.check_health()
@@ -211,12 +209,34 @@ class PetController(QObject):
except Exception as exc:
self.log.emit(f"Server not reachable yet ({exc}) — will keep trying per-request.")
self._start_notification_bridge()
self._report_self_restart()
self._guarded(self._report_self_restart, "restart report")
self._loop()
except Exception as exc:
self.log.emit(f"Pipeline stopped unexpectedly: {exc!r}")
finally:
if self._notification_watcher is not None:
self._notification_watcher.stop()
self.finished.emit()
def _guarded(self, work, label: str) -> bool:
"""Run *work*, absorbing anything it raises.
The pipeline is one thread driving a state machine that raises on an
illegal transition (deliberately see state.py), plus a dozen
best-effort subsystems that shell out, hit the network, or touch the
filesystem. Any one of them raising something unforeseen used to end the
whole session. Here, it costs a log line and a forced return to IDLE,
which is the only state it's always safe to resume from.
Returns True if *work* completed without raising."""
try:
work()
return True
except Exception as exc:
self.log.emit(f"Recovered from a {label} failure: {exc!r}")
self._state.force(PetState.IDLE)
return False
def _loop(self) -> None:
while self._running:
if self._muted:
@@ -231,7 +251,7 @@ class PetController(QObject):
if not self._running:
return
continue
self._handle_conversation_turn()
self._guarded(self._handle_conversation_turn, "conversation turn")
def _wait_for_wake_or_click(self) -> bool:
"""True once either the wake phrase was heard or a click-to-talk
@@ -243,7 +263,12 @@ class PetController(QObject):
self._stream,
should_continue=should_continue,
threshold=self.wake_threshold, # callable: the tuner slider is live
on_tick=self._maybe_heartbeat,
# Guarded: on_tick is the one place control returns to us during a
# listen that can block for minutes, and everything it drives
# (update check, nap probe, notification forwarding) touches the
# network or shells out. Unguarded, any of them raising would unwind
# the listen loop and end the session.
on_tick=lambda: self._guarded(self._maybe_heartbeat, "heartbeat"),
on_score=self._observe_wake_score,
)
if not self._running:
@@ -266,28 +291,21 @@ class PetController(QObject):
self._follow_ups = 0
self._state.transition(PetState.LISTENING)
# Transcribe while they talk rather than after: frames go to Deepgram
# as they are captured, so the text is ready the moment the VAD says
# they stopped. None here just means the one-shot path will do it.
streamed = stt_stream.StreamingTranscriber.open()
pcm = mic.record_utterance(
self._stream,
should_continue=self._should_continue,
# Answering a question deserves longer than saying the wake word
# on purpose does — you were just asked something.
grace_s=config.FOLLOW_UP_GRACE_SECONDS if following_up else None,
on_frame=streamed.feed if streamed is not None else None,
)
if pcm is None:
if streamed is not None:
streamed.finish()
self._follow_ups = 0 # silence ends the chain
self._state.transition(PetState.IDLE)
return
self._state.transition(PetState.THINKING)
try:
text = self._transcribe(pcm, streamed)
text = stt.transcribe(pcm)
except stt.SttError as exc:
self.log.emit(f"STT failed: {exc}")
self._state.transition(PetState.ERROR)
@@ -299,10 +317,19 @@ class PetController(QObject):
self.log.emit(f"You: {text}")
self.history.add(history_mod.USER, text, time.time())
# "stop", "come here", "say that again" — answered here, without the
# round trip. Never on a follow-up turn: Bolt asked you something and
# the answer is his, even if it happens to look like a body command.
if not following_up and self._handle_local_intent(text):
self._state.transition(PetState.IDLE)
return
try:
# What's focused right now rides along, so "what's this error?"
# has a referent without you having to describe the window.
reply = self._ask_server(self._with_context(text))
reply = server_client.converse(
self._with_context(text), on_command=self._handle_command
)
except server_client.ServerError as exc:
self.log.emit(f"Server error: {exc}")
self._state.transition(PetState.ERROR)
@@ -311,33 +338,63 @@ class PetController(QObject):
self._check_deliveries()
self._apply_voice(reply)
if reply.spoken:
# Streamed: every sentence was spoken and logged as it arrived.
# The text still matters — a reply ending on a question should keep
# the mic open — but saying it again would repeat the whole answer.
self._after_speaking(reply.text, completed=True)
else:
self._speak(reply.text)
self._state.transition(PetState.IDLE)
self._maybe_self_restart()
def _transcribe(self, pcm, streamed) -> str:
"""The transcript, from the live stream if it produced one.
def _handle_local_intent(self, text: str) -> bool:
"""Answer *text* locally if it's one of the closed set of body commands
in intents.py. Returns True if it was handled (no server call).
The fallback is not a rare path to be tolerated it is the safety net
that lets streaming be switched on at all. Whatever happened to the
socket, the full audio is still buffered here, so a failed stream costs
one ordinary upload and nothing else."""
if streamed is not None:
try:
text = streamed.finish()
except Exception:
self.log.emit("Streaming STT failed — falling back.")
text = ""
if text:
self.log.emit("(transcribed while you spoke)")
return text
return stt.transcribe(pcm)
The effects live here rather than in intents.py for the same reason
pet_actions splits parse from describe: recognising the phrase is pure
and testable, doing the thing needs the controller's state, the tray's
nap override and a Qt signal to the window."""
if not config.LOCAL_INTENTS:
return False
intent = intents_mod.recognize(text)
if intent is None:
return False
self.log.emit(f"Local intent: {intent.name} (answered without the server)")
if intent.name == "stop":
# Nothing to say and nothing to do: silence is the acknowledgement.
# Also ends any follow-up chain — "never mind" means the
# conversation is over, not that we should keep the mic open.
self._follow_ups = 0
self._pending_follow_up = False
self._talk_now.clear()
return True
if intent.name == "repeat":
last = self.history.last(history_mod.PET)
if last is None:
self._speak("I haven't said anything yet.")
else:
# remember=False: replaying a line isn't a new turn. Appending it
# would make "say that again" twice over read back as a
# conversation where Bolt volunteered the same thing three times.
self._speak(last.text, remember=False)
return True
if intent.name == "voice_reset":
had_voice = bool(self._voice_id)
self.reset_voice()
self._speak(intent.speak if had_voice else "That is my normal voice.")
return True
action = intent.action
if action is not None:
if action.get("action") == "nap":
# Through set_napping, not just the signal, so a spoken "go to
# sleep" overrides the quiet-hours schedule exactly like the
# tray's Nap entry and `petctl nap` do — otherwise the next
# schedule check would undo it within ten seconds.
self.set_napping(bool(action["enabled"]))
self.action.emit(dict(action))
if intent.speak:
self._speak(intent.speak)
return True
def _with_context(self, text: str) -> str:
"""Everything the server gets alongside what you actually said: the
@@ -365,9 +422,26 @@ class PetController(QObject):
self._pet_monitor = int(index)
def _handle_command(self, command: str) -> str:
"""Server-relayed command. `petctl ...` drives the pet's body and
`filectl ...` does local file read/write/edit neither ever reaches
a shell; everything else is a real command, exactly as before (see
"""Server-relayed command, with the guarantee the relay depends on: this
always returns a string.
The server is blocked on `/desk/tool_result` while this runs. If it
raises instead of answering, the relay never posts, the turn dies
mid-flight, and the server sits out its own timeout on a conversation it
can't finish — the worst available failure mode, because it's silent on
both ends. Handing the exception back as command output instead means
Bolt can read what went wrong and say so, or try something else, inside
the same turn."""
try:
return self._dispatch_command(command)
except Exception as exc:
self.log.emit(f"Command handler failed: {exc!r}")
return f"[error] the pet couldn't run that: {exc}"
def _dispatch_command(self, command: str) -> str:
"""`petctl ...` drives the pet's body, `dialoguectl ...` plays a scene
and `filectl ...` does local file read/write/edit none of them ever
reach a shell; everything else is a real command, exactly as before (see
the security notes in the README)."""
try:
action = pet_actions.parse(command)
@@ -538,26 +612,6 @@ class PetController(QObject):
self._speak(reply.text)
self._state.force(PetState.IDLE)
def _ask_server(self, text: str):
"""One turn with the server, streamed when possible.
Streaming speaks each sentence as it is generated, so the wait is
time-to-first-sentence rather than the whole model call. It falls back
to the ordinary request/response path when the server has no streaming
endpoint, or when a stream dies *before* anything was spoken after
that, retrying would say the first half twice."""
if config.STREAMING_REPLIES:
try:
return server_client.converse_stream(
text, on_say=self._speak_stream_chunk,
on_command=self._handle_command,
)
except server_client.ServerError as exc:
self.log.emit(f"Streaming unavailable ({exc}) — using the plain path.")
return server_client.converse(
text, on_command=self._handle_command, on_say=self._speak_holding,
)
def _play_dialogue(self, scene: dict) -> str:
"""Play a `dialoguectl` scene and report back up the relay.
@@ -591,8 +645,20 @@ class PetController(QObject):
# scene or fix the voice, and try again inside the same turn.
return f"[dialogue] couldn't synthesize it: {exc}"
completed = self._speaker.say_pcm(
speech.Utterance.scene(text), pcm=pcm, sample_rate=sample_rate)
resume = self._state.state
self._state.transition(PetState.TALKING)
self.said.emit(speech_text.for_display(text))
self.history.add(history_mod.PET, text, time.time())
should_stop = None
if self._barge_in is not None:
self._barge_in.reset()
should_stop = self._barge_in.check
completed = tts.play_pcm(pcm, sample_rate, should_stop=should_stop)
if self._barge_in is not None:
self._barge_in.reset() # the pet's own voices are in the wake window
if resume in (PetState.THINKING, PetState.IDLE):
self._state.transition(resume)
if not completed:
return dialogue_mod.describe(scene) + " (interrupted — they talked over it)"
@@ -622,31 +688,34 @@ class PetController(QObject):
elif not config.VOICE_STICKY:
self.reset_voice()
def _speak(self, text: str) -> None:
"""The answer: transcript, bubble, and the follow-up rule."""
completed = self._speaker.say(speech.Utterance.reply(text))
self._after_speaking(text, completed=completed,
detail=self._speaker.last_detail)
def _speak(self, text: str, remember: bool = True) -> None:
self._state.transition(PetState.TALKING)
# Bubble gets the markdown stripped but emoji kept (it can't render
# **bold** but draws emoji fine); tts.speak() does its own, stricter
# sanitizing for the voice.
self.said.emit(speech_text.for_display(text))
self.log.emit(f"Bolt: {text}")
if remember:
self.history.add(history_mod.PET, text, time.time())
def _speak_holding(self, text: str) -> None:
""""Give me a sec" while a tool runs — filler, so no transcript, and it
resumes the state it interrupted because the turn isn't over."""
self._speaker.say(speech.Utterance.holding(text))
def _speak_stream_chunk(self, text: str) -> None:
"""One sentence of a streamed answer, spoken the moment it arrives."""
completed = self._speaker.say(speech.Utterance.stream_chunk(text))
if not completed:
self.log.emit("Interrupted — listening.")
self._follow_ups = 0
self._talk_now.set()
def _after_speaking(self, text: str, *, completed: bool, detail: str = "") -> None:
"""What happens once an answer has been said, however it was said.
Shared by the plain and streamed paths: a streamed reply is spoken
sentence by sentence, but it still has to obey the same rules about
keeping the mic open when it ended on a question."""
should_stop = None
if self._barge_in is not None:
self._barge_in.reset()
should_stop = self._barge_in.check
completed = tts.speak(
text,
on_error=lambda exc: self.log.emit(f"TTS failed: {exc}"),
should_stop=should_stop,
voice_id=self._voice_id or None,
)
# Read the scoring history *before* resetting, or the log reports the
# blank counters instead of what actually fired.
detail = self._barge_in_detail()
if self._barge_in is not None:
# Playback fed the pet's own voice into the wake model's rolling
# window. Clear it before the idle listener starts scoring again,
# or Bolt's last sentence is still in there being re-scored.
self._barge_in.reset()
if not completed:
# You talked over it — take that as the start of the next turn
# rather than making you say the wake word again.
@@ -660,28 +729,35 @@ class PetController(QObject):
f"Asked a question — listening for your answer "
f"({self._follow_ups}{'/' + str(cap) if cap > 0 else ''})."
)
# The tail of the reply we just played is still in the mic's ring
# buffer, and we're about to start recording with a VAD that will
# take it for the start of your answer — Bolt's own last words,
# transcribed and sent back to him as if you'd said them. Nothing you
# said can be in there: playback ran to completion, so if you had
# spoken, barge-in would have cut it and taken the other branch.
dropped = mic.flush(self._stream)
if dropped:
self.log.emit(f"Dropped {dropped} buffered frames of my own voice.")
self._pending_follow_up = True
self._talk_now.set()
def _should_follow_up(self, text: str) -> bool:
"""Whether *text* leaves the pet waiting on an answer.
The rule itself question, cap, off switch is
`speech.follow_up_decision`, because it is a rule with an off-by-one in
it and deserves a test that needs no audio. What stays here is the part
that is genuinely the controller's: mute, and the log line. Muted is
excluded because mute means "don't listen to me" and an automatic turn
would walk straight past it. Napping isn't: quiet hours suppress the
pet *starting* something, and a question is only ever asked in reply to
you."""
if self._muted:
Muted is excluded because mute means "don't listen to me" an
automatic turn would walk straight past it. Napping isn't: quiet
hours suppress the pet *starting* something, and a question is only
ever asked in reply to you."""
if not config.FOLLOW_UP_LISTEN or self._muted:
return False
if not speech_text.is_question(text):
return False
keep, why = speech.follow_up_decision(text, completed=True, follow_ups=self._follow_ups)
if not keep and "cap" in why:
# Only worth mentioning the cap on a reply that would otherwise have
# kept listening, or it fires on every statement the pet makes.
if config.FOLLOW_UP_MAX_TURNS > 0 and self._follow_ups >= config.FOLLOW_UP_MAX_TURNS:
self.log.emit("Follow-up limit reached — say the wake word to keep going.")
return keep
return False
return True
def _barge_in_detail(self) -> str:
"""Why the interruption fired, for the log. How far into playback it
@@ -739,55 +815,70 @@ class PetController(QObject):
def _queue_notification(self, notification: notifications.Notification) -> None:
"""Called on the watcher thread — just queue it; forwarding happens on
the pipeline thread where it can't collide with a live conversation.
Only the *filter* applies here. The rate limit used to as well, which
meant a second message arriving inside the window was silently thrown
away with NOTIFICATION_MIN_INTERVAL_SECONDS=60, two texts a minute
apart and you only ever heard about one of them. Losing a message from
a person to save a round trip is the wrong trade; they are batched at
the far end instead, which costs the same one round trip and keeps
them all."""
if not self._notification_gate.matches(notification):
the pipeline thread where it can't collide with a live conversation."""
now = time.monotonic()
if not self._notification_gate.should_forward(notification, now):
return
with self._notification_lock:
if len(self._pending_notifications) >= _MAX_PENDING_NOTIFICATIONS:
self._pending_notifications.pop(0) # bound it; oldest goes first
self._pending_notifications.append(notification)
if len(self._pending_notifications) == self._pending_notifications.maxlen:
# Say so rather than dropping in silence: a full queue means the
# bridge is matching more than the pet can plausibly speak, and
# the filter is what wants tightening.
self.log.emit("Notification queue full — dropping the oldest.")
self._pending_notifications.append((now, notification))
def _drain_notifications(self) -> None:
"""Forward everything waiting as ONE turn.
Batching is what makes it safe to keep every notification: five that
arrived while the pet was mid-conversation become one message and one
round trip, instead of five separate interruptions queued up to fire
back to back."""
with self._notification_lock:
pending, self._pending_notifications = self._pending_notifications, []
if not pending or not self._running or self._napping:
pending = list(self._pending_notifications)
self._pending_notifications.clear()
now = time.monotonic()
max_age = config.NOTIFICATION_MAX_AGE_SECONDS
if max_age > 0:
fresh = [entry for entry in pending if now - entry[0] <= max_age]
if len(fresh) != len(pending):
self.log.emit(
f"Skipping {len(pending) - len(fresh)} notification(s) older than "
f"{int(max_age)}s."
)
pending = fresh
for index, (_stamped, notification) in enumerate(pending):
if not self._running or self._napping:
# Put back what we haven't forwarded — the old code swapped the
# queue out and then returned, silently dropping the remainder
# the moment a nap started mid-drain.
self._requeue_notifications(pending[index:])
return
for notification in pending:
self.log.emit(f"Notification: {notification.as_text()}")
self.history.add(history_mod.SYSTEM, notification.as_text(), time.time())
if len(pending) == 1:
message = f"[desktop notification] {pending[0].as_text()}"
else:
lines = "\n".join(f"- {n.as_text()}" for n in pending)
message = f"[{len(pending)} desktop notifications]\n{lines}"
try:
reply = server_client.converse(
message, on_command=self._handle_command, on_say=self._speak_holding,
f"[desktop notification] {notification.as_text()}",
on_command=self._handle_command,
)
except server_client.ServerError as exc:
# Keep this one and everything behind it for the next heartbeat:
# the server being briefly down shouldn't silently eat the
# backlog. The age limit is what stops that retrying forever.
self.log.emit(f"Couldn't forward notification: {exc}")
self._requeue_notifications(pending[index:])
return
self._check_deliveries()
self._apply_voice(reply)
if reply.text.strip() and not reply.spoken:
if reply.text.strip():
self._speak(reply.text)
self._state.transition(PetState.IDLE)
def _requeue_notifications(self, entries: list) -> None:
"""Push undelivered notifications back on the front, oldest first, so a
retry keeps their original order (and their original timestamps, so a
retry loop can't keep a stale one alive indefinitely)."""
if not entries:
return
with self._notification_lock:
self._pending_notifications.extendleft(reversed(entries))
# ── file delivery ────────────────────────────────────────────────────
def _check_deliveries(self) -> None:
@@ -859,36 +950,20 @@ class PetController(QObject):
# ── heartbeat ────────────────────────────────────────────────────────
def _maybe_heartbeat(self) -> None:
"""Called from the wake listener's tick (~every WAKE_CHECK_INTERVAL_SECONDS).
Two different cadences live here, and conflating them was costing
minutes. A *notification* is an event that already happened it should
go out as soon as the pet is free, which is the next tick. The
*heartbeat* is a poll, and polling the server every 1.2s would be
absurd, so it stays on its own interval."""
self._refresh_nap_state()
self._maybe_update()
if self._update_pending:
return # on the way out — don't start a conversation now
# Notifications: every tick, not every heartbeat.
if self._state.state == PetState.IDLE and not self._napping:
self._drain_notifications()
now = time.monotonic()
if now - self._last_heartbeat < config.HEARTBEAT_INTERVAL_SECONDS:
return
self._last_heartbeat = now
if self._state.state != PetState.IDLE:
# Mid-conversation. Do NOT stamp the clock — an earlier version
# did, so a heartbeat that landed while the pet was talking burned
# its slot and waited another full interval. With follow-up
# listening that could repeat for several cycles, which is why a
# notification could sit unspoken for five or ten minutes.
return
if self._napping:
return # quiet hours: still answers when spoken to, just doesn't start
self._last_heartbeat = now
self._check_deliveries()
self._drain_notifications()
if self._state.state != PetState.IDLE:
return
try:
-8
View File
@@ -224,12 +224,4 @@ def describe(action: dict, *, played: bool = True) -> str:
return (
f"[dialogue] played {len(lines)} line{'s' if len(lines) != 1 else ''} "
f"in {len(voices)} voice{'s' if len(voices) != 1 else ''}: {', '.join(voices)}{note}"
# The scene was spoken out loud before this result got back to the
# server, and the model has no other way to know that. Without saying
# so it writes a final reply summarising what the user just heard, and
# the pet says the same thing twice in a row (observed 2026-07-31).
# An instruction delivered here, at the moment it applies, lands far
# better than a rule buried in a long system prompt.
"\nThe user HEARD this already. Do not repeat, summarise or narrate it "
"in your reply — answer with at most one short line, or nothing new."
)
-272
View File
@@ -1,272 +0,0 @@
"""`python -m bolt_pet --doctor` — is this install actually going to work?
Written after a week of debugging things a preflight would have shown in one
command: a missing port publish, a wrong reverse-proxy header, a venv whose
python was a zero-byte file, an OCR engine that was never installed. Every one
of those presented as "the pet is being weird" and took a conversation to find.
Each check is independent and reports one of three things ok, a warning
(works, but degraded), or a failure (this will not do what you expect) and
says *what to do about it* rather than just what is wrong. Nothing here raises:
a doctor that crashes on a broken install is diagnosing the wrong patient.
Deliberately does not open the mic or call a paid API by default. It checks
that the door is unlocked, not that the room is furnished; `--deep` is there
when you want it to actually knock.
"""
from __future__ import annotations
import importlib
import shutil
import sys
from dataclasses import dataclass
from typing import Callable, Optional
from . import config
OK, WARN, FAIL = "ok", "warn", "fail"
_MARKS = {OK: " ok ", WARN: " warn ", FAIL: " FAIL "}
@dataclass
class Check:
name: str
status: str
detail: str = ""
fix: str = ""
def line(self) -> str:
text = f"[{_MARKS[self.status]}] {self.name:22} {self.detail}"
if self.fix and self.status != OK:
text += f"\n{'':32}{self.fix}"
return text
def _module(name: str) -> bool:
try:
importlib.import_module(name)
return True
except Exception:
return False
# ── the checks ──────────────────────────────────────────────────────────────
def check_config() -> Check:
missing = config.missing_config()
if missing:
return Check("server config", FAIL, f"missing {', '.join(missing)}",
"set them in .env — without these the controller exits at startup")
return Check("server config", OK, f"{config.SERVER_URL} as {config.SESSION_ID}")
def check_server(deep: bool = False) -> Check:
if not config.SERVER_URL:
return Check("server", FAIL, "no BOLT_SERVER_URL",
"cp .env.example .env and set BOLT_SERVER_URL to your Bolt server")
if not deep:
return Check("server", OK, f"{config.SERVER_URL} (not contacted; --deep to try)")
try:
from . import server_client
health = server_client.check_health(timeout=8)
return Check("server", OK, f"reachable — {health}")
except Exception as exc:
return Check("server", FAIL, f"unreachable: {exc}",
"check the URL, the key, and that the container is up")
def check_streaming_endpoint(deep: bool = False) -> Check:
"""The streamed-reply endpoint is newer than some deployed servers."""
if not config.STREAMING_REPLIES:
return Check("streamed replies", WARN, "disabled (STREAMING_REPLIES=false)")
if not deep:
return Check("streamed replies", OK, "enabled (endpoint not probed)")
try:
import requests
response = requests.post(
f"{config.SERVER_URL}/desk/converse_stream",
json={"session_id": config.SESSION_ID, "text": ""},
headers={"X-Desk-Api-Key": config.API_KEY}, timeout=8, stream=True,
)
if response.status_code == 404:
return Check("streamed replies", WARN, "server has no /desk/converse_stream",
"update the server, or set STREAMING_REPLIES=false to skip the probe")
return Check("streamed replies", OK, f"endpoint answered {response.status_code}")
except Exception as exc:
return Check("streamed replies", WARN, f"probe failed: {exc}")
def check_microphone(deep: bool = False) -> Check:
if not _module("sounddevice"):
return Check("microphone", FAIL, "sounddevice is not installed",
"pip install -r requirements.txt")
try:
import sounddevice as sd
devices = [d for d in sd.query_devices() if d.get("max_input_channels", 0) > 0]
if not devices:
return Check("microphone", FAIL, "no input devices",
"on Linux check PipeWire/PulseAudio is running as your user")
chosen = config.MIC_DEVICE or "system default"
if not deep:
return Check("microphone", OK, f"{len(devices)} input device(s), using {chosen}")
with sd.InputStream(samplerate=config.SAMPLE_RATE, channels=1, dtype="int16",
device=config.MIC_DEVICE, blocksize=config.FRAME_LEN):
pass
return Check("microphone", OK, f"opened at {config.SAMPLE_RATE} Hz ({chosen})")
except Exception as exc:
return Check("microphone", FAIL, f"could not open: {exc}",
"don't run the pet as root — PortAudio can't reach your PipeWire socket")
def check_wake_model() -> Check:
from pathlib import Path
path = Path(config.WAKE_MODEL_PATH)
if not path.exists():
return Check("wake word", FAIL, f"{path.name} is missing",
"it ships in the project root; check WAKE_MODEL_FILE")
if not _module("openwakeword"):
return Check("wake word", FAIL, "openwakeword is not installed",
"pip install -r requirements.txt")
return Check("wake word", OK, f"{path.name}, threshold {config.WAKE_WORD_THRESHOLD}")
def check_stt() -> Check:
if not config.DEEPGRAM_API_KEY:
return Check("speech-to-text", FAIL, "no DEEPGRAM_API_KEY",
"set it in .env — nothing you say can be transcribed without it")
if config.STT_STREAMING and not _module("websocket"):
return Check("speech-to-text", WARN, "streaming on, but websocket-client is missing",
"pip install websocket-client — it falls back to one-shot uploads")
mode = "streaming" if config.STT_STREAMING else "one-shot"
return Check("speech-to-text", OK, f"Deepgram {config.DEEPGRAM_MODEL}, {mode}")
def check_tts() -> Check:
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
if _module("pyttsx3"):
return Check("text-to-speech", WARN, "no ElevenLabs key/voice — offline voice only",
"set ELEVENLABS_API_KEY and ELEVENLABS_VOICE_ID for the real voice")
return Check("text-to-speech", FAIL, "no ElevenLabs config and no pyttsx3 fallback")
return Check("text-to-speech", OK,
f"ElevenLabs {config.ELEVENLABS_MODEL_ID}, voice …{config.ELEVENLABS_VOICE_ID[-6:]}")
def check_dialogue() -> Check:
if not config.DIALOGUE:
return Check("multi-voice scenes", WARN, "disabled (DIALOGUE=false)")
from . import dialogue
cast = dialogue.parse_voice_map(config.DIALOGUE_VOICES)
if not cast:
return Check("multi-voice scenes", WARN, "no cast configured — only 'self' works",
'set DIALOGUE_VOICES=narrator:<id>,villain:<id>')
return Check("multi-voice scenes", OK, f"{len(cast)} voice(s): {', '.join(sorted(cast))}")
def check_screen_text() -> Check:
if not config.SCREEN_TEXT:
return Check("screen reading", WARN, "disabled (SCREEN_TEXT=false)")
if not _module("mss"):
return Check("screen reading", WARN, "mss is not installed — petctl read will decline",
"pip install mss (and note it cannot capture on Wayland)")
engine = "pytesseract" if _module("pytesseract") else (
"rapidocr" if _module("rapidocr_onnxruntime") else "")
if not engine:
return Check("screen reading", WARN, "no OCR engine",
"pip install pytesseract && apt install tesseract-ocr, "
"or pip install rapidocr-onnxruntime")
if engine == "pytesseract" and not shutil.which("tesseract"):
return Check("screen reading", WARN, "pytesseract is installed but tesseract is not",
"apt install tesseract-ocr")
return Check("screen reading", OK, f"mss + {engine}")
def check_hotkey() -> Check:
if not _module("pynput"):
return Check("push-to-talk", WARN, "pynput is not installed",
"the wake word still works; pip install pynput for the hotkey")
import os
if os.environ.get("WAYLAND_DISPLAY") and not os.environ.get("DISPLAY"):
return Check("push-to-talk", WARN, "Wayland session — global hotkeys usually blocked",
"use the wake word, or click the pet")
return Check("push-to-talk", OK, config.PUSH_TO_TALK_HOTKEY)
def check_sprites() -> Check:
from pathlib import Path
root = Path(__file__).resolve().parent / "assets" / "sprites"
if not root.exists():
return Check("sprites", WARN, "no art — the placeholder blob will be drawn",
"python scripts/generate_bolt_sprites.py")
counts = {d.name: len(list(d.glob("*.png"))) for d in sorted(root.iterdir()) if d.is_dir()}
empty = [name for name, count in counts.items() if count == 0]
if empty:
return Check("sprites", WARN, f"no frames for: {', '.join(empty)}",
"python scripts/generate_bolt_sprites.py")
return Check("sprites", OK, ", ".join(f"{n} {c}" for n, c in counts.items()))
def check_latency() -> Check:
"""The setting most likely to make it feel slow, and the least obvious."""
silence = config.SILENCE_END_SEC
if silence >= 1.5:
return Check("turn latency", WARN, f"VAD_SILENCE_END_SEC={silence:g}s of dead air per turn",
"0.8-1.0 feels markedly snappier; it is pure wait before anything starts")
return Check("turn latency", OK, f"silence timeout {silence:g}s, "
f"streaming {'on' if config.STREAMING_REPLIES else 'off'}")
CHECKS: tuple[tuple[str, Callable], ...] = (
("config", check_config),
("server", check_server),
("stream", check_streaming_endpoint),
("mic", check_microphone),
("wake", check_wake_model),
("stt", check_stt),
("tts", check_tts),
("dialogue", check_dialogue),
("screen", check_screen_text),
("hotkey", check_hotkey),
("sprites", check_sprites),
("latency", check_latency),
)
def run(deep: bool = False) -> list[Check]:
results = []
for _name, check in CHECKS:
try:
try:
results.append(check(deep))
except TypeError:
results.append(check())
except Exception as exc: # a broken check must not hide the others
results.append(Check(_name, FAIL, f"the check itself failed: {exc}"))
return results
def main(argv: Optional[list] = None) -> int:
argv = list(argv if argv is not None else sys.argv[1:])
deep = "--deep" in argv
print(f"Bolt pet preflight{' (deep: contacting the server and opening the mic)' if deep else ''}\n")
results = run(deep=deep)
for check in results:
print(check.line())
failures = [c for c in results if c.status == FAIL]
warnings = [c for c in results if c.status == WARN]
print()
if failures:
print(f"{len(failures)} problem(s) will stop this working. Fix those first.")
elif warnings:
print(f"Ready. {len(warnings)} thing(s) degraded but working.")
else:
print("Everything checks out.")
return 1 if failures else 0
+206
View File
@@ -0,0 +1,206 @@
"""Things you say to the pet that the server has no business answering.
"stop", "come here", "go to sleep", "say that again", "use your normal voice"
none of these are questions for Bolt's brain. They're commands to the *body*,
and today every one of them costs a full turn: Deepgram, a `/desk/converse`
round trip, a model deciding to emit `petctl`, then ElevenLabs. Two to four
seconds and three network hops to make the pet walk left, and it only works at
all if the server's prompt happens to advertise the right verb — which is
exactly why `petctl voice reset` needs a block in the server's pet prompt (see
CLAUDE.md) or the model never emits it. Recognising the phrase here removes
both the latency and that coupling: "go back to your normal voice" works
whether or not the server was ever told the voice can be reset.
The whole design problem is **not stealing real requests**. Three rules keep
it honest:
1. **Whole-utterance, exact match after normalisation.** Never substring. So
"stop" is an intent and "stop the docker container" is a question for the
server the distinction a substring match would destroy.
2. **The phrase table is closed and small.** Every entry is something with no
plausible reading as a request for Bolt to *do work*. Anything arguable
("no thanks", "nothing") is deliberately absent see rule 3 for why a
wrong guess is expensive.
3. **Nothing is recognised mid-conversation.** The controller skips this
entirely on a follow-up turn: if Bolt just asked you something, your answer
belongs to him, and swallowing "never mind" locally would leave the server
holding a question it never got an answer to. Local intents are only ever
for turns *you* started.
Both sides of the comparison go through `normalize()` the table is
canonicalised at import so phrases can be written the way a person says them
("go back to your normal voice") without every variant having to be spelled
out. Filler is dropped from anywhere, not just the ends, because STT scatters
it ("hey bolt, could you please just stop now").
Pure classification, like pet_actions.parse: this module decides *what was
meant* and hands back an action in the same shape pet_actions produces, so
`controller.action` and `PetWindow.apply_action` need no new vocabulary. The
effects live in controller._handle_local_intent.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Optional
# Words with no bearing on any command in the table, dropped wherever they
# appear. Kept deliberately short: every entry here is a word that can't
# distinguish one of these phrases from another, and adding one that can is how
# two intents quietly collide (the builder below raises if that happens).
_FILLER = frozenset({
"the", "a", "an", "my", "your", "yours", "its", "to", "of", "and",
"please", "just", "that", "some", "bolt", "thunderbolt", "pet", "buddy",
})
# Dropped only from the front — the politeness/address ramp STT reliably
# prefixes. Not safe to drop mid-phrase (a bare "do" or "go" carries meaning
# elsewhere), which is why this is separate from _FILLER.
_LEADING_FILLER = frozenset({
"hey", "hi", "hello", "yo", "ok", "okay", "um", "uh", "er", "so",
"can", "could", "would", "will", "you", "i", "id", "like", "lets",
"let", "us", "do", "go", "then", "now",
})
_TRAILING_FILLER = frozenset({
"ok", "okay", "thanks", "thank", "you", "boy", "already", "now",
})
_KEEP = re.compile(r"[^a-z0-9 ]+")
def normalize(text: str) -> str:
"""Reduce an utterance to the bare command, or "" if nothing is left.
Lowercase, punctuation stripped (STT punctuates inconsistently), filler
dropped. Not a stemmer and deliberately not clever its only job is to
make the same command spoken two ways land on the same string, without
ever turning one command into a different one."""
words = [word for word in _KEEP.sub(" ", (text or "").lower()).split()
if word not in _FILLER]
while words and words[0] in _LEADING_FILLER:
words.pop(0)
while words and words[-1] in _TRAILING_FILLER:
words.pop()
return " ".join(words)
@dataclass(frozen=True)
class Intent:
"""One recognised local command.
*action* is a pet_actions-shaped dict for the UI (or None when there's
nothing for the body to do); *speak* is what to say out loud, empty for the
intents where doing the thing silently *is* the acknowledgement the pet
visibly moves, and a spoken confirmation would only make it slower. "stop"
in particular has to be silent: answering "okay!" when told to be quiet is
a comedy sketch, not a feature.
"""
name: str
action: Optional[dict] = None
speak: str = ""
# Intent -> (the Intent, the phrases that mean it, written as spoken).
_TABLE: tuple[tuple[Intent, tuple[str, ...]], ...] = (
(
Intent("stop"),
("stop", "stop talking", "stop it", "be quiet", "quiet", "shut up",
"hush", "never mind", "nevermind", "forget it", "cancel",
"cancel that", "drop it", "enough"),
),
(
Intent("nap", {"action": "nap", "enabled": True}, "Night."),
("go to sleep", "take a nap", "have a nap", "go to bed", "bedtime",
"goodnight", "good night", "get some rest"),
),
(
Intent("wake", {"action": "nap", "enabled": False}, "I'm up."),
("wake up", "get up", "rise and shine", "you're awake", "are you awake"),
),
(
Intent("come", {"action": "move", "anchor": "cursor"}),
("come here", "come to me", "come back", "over here", "follow me",
"follow my cursor"),
),
(
Intent("go_away", {"action": "move", "anchor": "bottom-right"}),
("go away", "move over", "move out of the way", "get out of the way",
"out of the way", "hide", "get lost", "shoo", "scram",
"go somewhere else"),
),
(
Intent("repeat"), # answered from history by the controller
("say that again", "say again", "repeat that", "repeat",
"what did you say", "what was that", "come again", "one more time",
"again", "sorry what"),
),
(
Intent("wander_on", {"action": "wander", "enabled": True}),
("go for a walk", "wander", "wander around", "walk around", "explore",
"stretch your legs", "roam"),
),
(
Intent("wander_off", {"action": "wander", "enabled": False}),
("stay still", "stay put", "stop moving", "stop wandering",
"don't move", "sit", "sit still", "stay", "settle down", "hold still"),
),
(
# Reachable from the server too (petctl voice reset), but only if its
# prompt mentions the verb. Recognising it here is what makes the
# phrase work regardless of what the server was told.
Intent("voice_reset", None, "Back to my own voice."),
("use your normal voice", "use your own voice", "your normal voice",
"go back to your normal voice", "be yourself", "be yourself again",
"stop doing that voice", "drop the voice", "talk normally",
"speak normally", "use your real voice"),
),
)
def _build() -> dict[str, Intent]:
"""Canonicalise the table, refusing to build an ambiguous one.
A phrase that normalises to "" would match an utterance of pure filler
("hey bolt"), and one that lands on the same string as a phrase from
another intent would silently bind to whichever was declared last. Both are
edit-time mistakes, so they fail at import rather than at 3am on a mic."""
table: dict[str, Intent] = {}
for intent, phrases in _TABLE:
for phrase in phrases:
key = normalize(phrase)
if not key:
raise ValueError(f"intent phrase {phrase!r} normalises to nothing")
existing = table.get(key)
if existing is not None and existing.name != intent.name:
raise ValueError(
f"phrase {phrase!r} ({key!r}) is claimed by both "
f"{existing.name} and {intent.name}"
)
table[key] = intent
return table
_BY_PHRASE = _build()
# Longest phrase in the table, in words. Anything longer can't match, so a real
# request skips normalisation entirely — this runs on every turn.
_MAX_WORDS = max(len(phrase.split()) for phrase in _BY_PHRASE)
def recognize(text: str) -> Optional[Intent]:
"""The intent *text* expresses, or None to send it to the server.
None is the safe answer and the common one: anything not matched verbatim
against the table belongs to Bolt."""
raw = (text or "").strip()
if not raw:
return None
# +6 words of slack for the filler about to be stripped ("hey bolt, could
# you please stop" is six words to reach a one-word command).
if len(raw.split()) > _MAX_WORDS + 6:
return None
intent = _BY_PHRASE.get(normalize(raw))
return intent
+65 -124
View File
@@ -19,7 +19,8 @@ Kept dependency-free beyond `requests` so it's easy to unit test with mocks.
from __future__ import annotations
import json
import os
import signal
import subprocess
from pathlib import Path
from typing import Callable, NamedTuple, Optional
@@ -30,6 +31,10 @@ from . import config, sudo_askpass
_MAX_RELAY_HOPS = 16
# Command output handed back up the relay is capped: it becomes part of the
# server's prompt, and a runaway `find /` would blow the context window.
_MAX_COMMAND_OUTPUT = 6000
class ServerError(Exception):
"""Raised when the server responds with an error payload or unreachable."""
@@ -44,10 +49,6 @@ class Reply(NamedTuple):
text: str
voice_id: str = ""
voice_name: str = ""
# True when the sentences were already spoken as they streamed in. The
# text is still carried — the follow-up rule needs to see whether the
# answer ended on a question — it just must not be read out again.
spoken: bool = False
def _headers() -> dict:
@@ -79,36 +80,72 @@ def run_local_command(command: str, timeout: int = None) -> str:
timeout = timeout or config.SUDO_COMMAND_TIMEOUT_SECONDS
timeout = timeout or config.COMMAND_TIMEOUT_SECONDS
try:
completed = subprocess.run(
command, shell=True, capture_output=True, text=True,
timeout=timeout, cwd=str(Path.home()), env=env,
# start_new_session puts the shell in its own process group so a timeout
# can kill the whole tree. subprocess.run() would only SIGKILL the `sh`
# itself, leaving whatever it spawned (a build, a `tail -f`, an ffmpeg)
# running forever with no parent watching — one relayed command that
# hangs shouldn't leak a process for the rest of the session.
process = subprocess.Popen(
command, shell=True, cwd=str(Path.home()), env=env, text=True,
stdout=subprocess.PIPE, stderr=subprocess.PIPE,
start_new_session=(os.name == "posix"),
)
output = (completed.stdout or "") + (completed.stderr or "")
return f"[exit {completed.returncode}]\n{output}"[:6000]
except subprocess.TimeoutExpired:
return f"[command timed out after {timeout}s]"
except Exception as exc:
return f"[command failed: {exc}]"
try:
stdout, stderr = process.communicate(timeout=timeout)
return _command_output(f"[exit {process.returncode}]", stdout, stderr)
except subprocess.TimeoutExpired:
stdout, stderr = _terminate(process)
# Whatever it managed to print before it hung is the useful part — a
# bare "timed out" tells the model nothing it can act on, and the last
# line of output usually says exactly what it was stuck waiting for.
return _command_output(f"[command timed out after {timeout}s]", stdout, stderr)
except Exception as exc:
_terminate(process)
return f"[command failed: {exc}]"
def _terminate(process: subprocess.Popen) -> tuple[str, str]:
"""Kill a timed-out command's whole process group and collect what it wrote.
SIGTERM first so a shell script can clean up, SIGKILL a moment later for
anything that ignores it. The final drain is itself time-boxed: a
grandchild holding the pipe open must not turn a timeout into a hang."""
try:
if os.name == "posix":
group = os.getpgid(process.pid)
os.killpg(group, signal.SIGTERM)
try:
process.wait(timeout=2)
except subprocess.TimeoutExpired:
os.killpg(group, signal.SIGKILL)
else:
process.kill()
except (ProcessLookupError, PermissionError, OSError):
pass # already gone, or never had its own group
try:
return process.communicate(timeout=2)
except Exception:
return "", ""
def _command_output(header: str, stdout: Optional[str], stderr: Optional[str]) -> str:
body = (stdout or "") + (stderr or "")
return f"{header}\n{body}"[:_MAX_COMMAND_OUTPUT]
def converse(
text: str,
on_command: Callable[[str], str] = run_local_command,
timeout: float = 120.0,
on_say: Optional[Callable[[str], None]] = None,
) -> Reply:
"""Send one turn of conversation to the desk API, relaying any commands
the server sends back until it produces a final reply.
*on_command* is injectable for tests; defaults to actually running the
command locally (matching bolt_desk.py's behavior).
*on_say* is called with a short holding line ("give me a sec") when the
server sends one alongside a command. It is the difference between silence
and an answer while a tool runs: the model's acknowledgement used to be
discarded server-side, so the whole round trip was dead air and the model
then repeated itself in the final reply. Optional, so an older pet against
a newer server simply stays quiet as before.
"""
headers = _headers()
try:
@@ -124,13 +161,6 @@ def converse(
for _ in range(_MAX_RELAY_HOPS):
if payload.get("type") != "command":
break
holding = str(payload.get("say") or "").strip()
if holding and on_say is not None:
# Spoken *before* the command runs — that is the whole point.
try:
on_say(holding)
except Exception:
pass # a failed acknowledgement must not cost the tool call
output = on_command(str(payload.get("command") or ""))
try:
response = requests.post(
@@ -152,103 +182,14 @@ def converse(
voice_id=str(payload.get("voice_id") or ""),
voice_name=str(payload.get("voice_name") or ""),
)
raise ServerError(str(payload.get("error") or "unknown server response"))
def converse_stream(
text: str,
on_say: Callable[[str], None],
on_command: Callable[[str], str] = run_local_command,
timeout: float = 180.0,
) -> Reply:
"""Same turn as converse(), but speaking each sentence as it arrives.
Without this the pet waits out the *entire* model call before a single
word is heard; with it the wait is time-to-first-sentence, which on a
multi-sentence answer is most of the difference.
Falls back by raising ServerError before anything has been spoken the
caller then retries the ordinary path and the user never finds out. Once a
sentence *has* been spoken there is no going back, so late failures end the
turn with whatever was said rather than repeating it.
"""
spoke_anything = False
try:
response = requests.post(
f"{config.SERVER_URL}/desk/converse_stream",
json={"session_id": config.SESSION_ID, "text": text},
headers=_headers(), timeout=timeout, stream=True,
)
response.raise_for_status()
for raw in response.iter_lines(decode_unicode=True):
if not raw:
continue
try:
event = json.loads(raw)
except (TypeError, ValueError):
continue
kind = str(event.get("type") or "")
if kind == "say":
line = str(event.get("text") or "").strip()
if line:
spoke_anything = True
on_say(line)
elif kind == "command":
# The tool loop is request/response, so the rest of the turn
# finishes through the ordinary relay rather than inside the
# stream — one protocol for tools, not two.
response.close()
return _finish_relay(event, on_command, on_say)
elif kind == "reply":
return Reply(
text=str(event.get("text") or ""),
voice_id=str(event.get("voice_id") or ""),
voice_name=str(event.get("voice_name") or ""),
spoken=bool(event.get("already_spoken")),
)
elif kind == "error":
raise ServerError(str(event.get("error") or "stream failed"))
except ServerError:
raise
except Exception as exc:
if spoke_anything:
# Half a reply is out loud already; ending quietly beats saying it
# all again through the fallback path.
return Reply(text="", spoken=True)
raise ServerError(f"streaming failed: {exc}") from exc
raise ServerError("stream ended without a reply")
def _finish_relay(event: dict, on_command, on_say) -> Reply:
"""Run the tool the stream handed over, then continue the classic relay."""
payload = dict(event)
headers = _headers()
for _ in range(_MAX_RELAY_HOPS):
if payload.get("type") != "command":
break
holding = str(payload.get("say") or "").strip()
if holding and on_say is not None:
try:
on_say(holding)
except Exception:
pass
output = on_command(str(payload.get("command") or ""))
try:
response = requests.post(
f"{config.SERVER_URL}/desk/tool_result",
json={"session_id": config.SESSION_ID,
"token": payload.get("token"), "output": output},
headers=headers, timeout=180,
)
payload = response.json()
except Exception as exc:
raise ServerError(f"couldn't reach the server during tool relay: {exc}") from exc
if payload.get("type") == "reply":
return Reply(
text=str(payload.get("text") or ""),
voice_id=str(payload.get("voice_id") or ""),
voice_name=str(payload.get("voice_name") or ""),
if payload.get("type") == "command":
# Fell out of the loop still being handed commands. Worth its own
# message: "unknown server response" sent everyone looking at the
# payload shape, when what actually happened is a model that kept
# calling tools and never answered.
raise ServerError(
f"the server kept relaying commands past the {_MAX_RELAY_HOPS}-hop cap "
"without producing a reply"
)
raise ServerError(str(payload.get("error") or "unknown server response"))
-194
View File
@@ -1,194 +0,0 @@
"""Everything the pet says, and the policy differences between kinds of saying.
There used to be four of these in controller.py a reply, a holding line, a
streamed sentence, a dialogue scene each written when its feature was built,
each repeating the same dance: transition state, show the bubble, maybe record
history, reset barge-in, call TTS, reset barge-in again, resume state, decide
whether to keep the mic open. Only the *policy* differed, and the copies had
already started to drift: one forgot to arm barge-in, another logged a
different prefix, a third skipped the follow-up rule.
So the dance lives here once, and the differences are data:
reply the answer. Transcript, bubble, follow-up rule, ends IDLE.
holding "give me a sec" while a tool runs. No transcript it is
filler, and the transcript should keep the answer. Resumes
whatever state it interrupted, because the turn isn't over.
stream one sentence of a streamed answer. Transcript and bubble like
a reply, but stays TALKING so the sprite doesn't flicker
between sentences, and the follow-up rule waits for the last.
scene a dialoguectl take. Transcript (the user heard it), resumes
mid-turn like a holding line.
The controller keeps the pipeline; this keeps the rules about talking.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Callable, Optional
from . import config, speech_text
from .state import PetState
@dataclass(frozen=True)
class Utterance:
"""One thing to say, and how saying it should behave."""
text: str
record: bool = True # goes in the transcript, or is it filler?
resume: bool = False # return to the state it interrupted (mid-turn)
hold_talking: bool = False # stay TALKING afterwards (more is coming)
follow_up: bool = True # may leave the mic open if it ends on a question
interruptible: bool = True # arm barge-in for this one
log_prefix: str = "Bolt"
@classmethod
def reply(cls, text: str) -> "Utterance":
return cls(text)
@classmethod
def holding(cls, text: str) -> "Utterance":
# Filler: no transcript, no follow-up, and not interruptible — cutting
# off "give me a sec" would strand the tool that is already running.
return cls(text, record=False, resume=True, follow_up=False,
interruptible=False, log_prefix="Bolt (holding)")
@classmethod
def stream_chunk(cls, text: str) -> "Utterance":
return cls(text, hold_talking=True, follow_up=False)
@classmethod
def scene(cls, text: str) -> "Utterance":
return cls(text, resume=True, follow_up=False)
class Speaker:
"""Says things on behalf of the controller.
Collaborators are passed in rather than reached for, so the whole of
speaking is testable without Qt, audio hardware or a state machine: hand it
fakes and assert on what came out."""
def __init__(self, *, state, tts, history=None, on_said=None, on_log=None,
barge_in: Callable[[], object] = lambda: None,
voice_id: Callable[[], str] = lambda: "",
detail_of: Callable[[], str] = lambda: "",
on_level: Optional[Callable[[float], None]] = None):
self._state = state
self._tts = tts
self._history = history
self._on_said = on_said or (lambda _text: None)
self._on_log = on_log or (lambda _msg: None)
# Read through a callable rather than held: the detector is built after
# the speaker (it needs the mic stream), replaced when barge-in mode
# changes, and swapped by tests. Two copies of it drifting apart is a
# bug nobody notices until the wake model starts hearing the pet.
self._barge_in_of = barge_in
self._voice_id = voice_id
# What actually fired, captured BEFORE the post-playback reset. Read it
# afterwards and every interruption reports 0.000 at frame 0 — which
# looks like hard evidence and is nothing of the sort.
self._detail_of = detail_of
self.last_detail = ""
self._on_level = on_level
@property
def _barge_in(self):
try:
return self._barge_in_of()
except Exception:
return None
def say(self, utterance: Utterance) -> bool:
"""Speak it. Returns False if it was interrupted.
The one place that knows the order these steps go in which is the
point, because getting that order wrong is invisible until the wake
model starts hearing the pet's own voice."""
line = speech_text.for_display(utterance.text)
if not line:
return True
resume_state = self._state.state
if self._state.state != PetState.TALKING:
self._state.transition(PetState.TALKING)
self._on_said(line)
self._on_log(f"{utterance.log_prefix}: {line}")
if utterance.record and self._history is not None:
self._history(utterance.text)
should_stop = None
if self._barge_in is not None and utterance.interruptible:
self._barge_in.reset()
should_stop = self._barge_in.check
completed = self._tts.speak(
utterance.text,
on_error=lambda exc: self._on_log(f"TTS failed: {exc}"),
should_stop=should_stop,
voice_id=self._voice_id() or None,
on_level=self._on_level,
)
self.last_detail = self._detail_of()
if self._barge_in is not None:
# Playback fed the pet's own voice into the wake model's rolling
# window. Clear it before the idle listener scores again, or the
# last sentence is still in there being re-heard.
self._barge_in.reset()
if self._on_level is not None:
self._on_level(0.0) # mouth closed; nothing is playing now
if utterance.resume and resume_state in (PetState.THINKING, PetState.IDLE):
self._state.transition(resume_state)
return bool(completed)
def say_pcm(self, utterance: Utterance, *, pcm, sample_rate: int) -> bool:
"""Speak audio that is already synthesized — a dialoguectl scene.
Same policy, same barge-in handling, same mouth; only the source of
the samples differs. Sharing this is why a scene can be talked over
exactly like an ordinary reply."""
line = speech_text.for_display(utterance.text)
resume_state = self._state.state
if self._state.state != PetState.TALKING:
self._state.transition(PetState.TALKING)
if line:
self._on_said(line)
self._on_log(f"{utterance.log_prefix}: {line}")
if utterance.record and self._history is not None:
self._history(utterance.text)
should_stop = None
if self._barge_in is not None and utterance.interruptible:
self._barge_in.reset()
should_stop = self._barge_in.check
completed = self._tts.play_pcm(
pcm, sample_rate, should_stop=should_stop, on_level=self._on_level)
self.last_detail = self._detail_of()
if self._barge_in is not None:
self._barge_in.reset()
if self._on_level is not None:
self._on_level(0.0)
if utterance.resume and resume_state in (PetState.THINKING, PetState.IDLE):
self._state.transition(resume_state)
return bool(completed)
def follow_up_decision(text: str, *, completed: bool, follow_ups: int) -> tuple[bool, str]:
"""Whether to keep listening after speaking, and why.
Split out as a pure function because it is a *rule* with an off-by-one cap
in it, and rules with counters deserve a test that doesn't need audio."""
if not completed:
return True, "interrupted"
if not speech_text.is_question(text):
return False, ""
cap = config.FOLLOW_UP_MAX_TURNS
if not config.FOLLOW_UP_LISTEN:
return False, ""
if cap > 0 and follow_ups >= cap:
return False, f"follow-up cap ({cap}) reached"
return True, "question"
+42 -42
View File
@@ -67,6 +67,27 @@ _SPOKEN_SYMBOLS = {
"=": " equals ",
}
# Abbreviations a voice spells out letter by letter ("eee gee") because the
# periods make them look like sentence boundaries. Written out instead — this
# has to run before _UNSPEAKABLE strips anything, and the trailing \.? keeps
# "etc" working with or without its period. Word-bounded so "vs" inside a
# filename is left alone.
_SPOKEN_ABBREVIATIONS = (
(re.compile(r"\be\.g\.?(?=\s|$)", re.IGNORECASE), "for example"),
(re.compile(r"\bi\.e\.?(?=\s|$)", re.IGNORECASE), "that is"),
(re.compile(r"\betc\.?(?=\s|$)", re.IGNORECASE), "and so on"),
(re.compile(r"\bvs\.?(?=\s|$)", re.IGNORECASE), "versus"),
(re.compile(r"\baka\b", re.IGNORECASE), "also known as"),
(re.compile(r"\bw/(?=\s)", re.IGNORECASE), "with"),
# "PR #42" -> "PR number 42"; a bare "#" is markup and _UNSPEAKABLE drops it.
(re.compile(r"#(?=\d)"), "number "),
# A long option's dashes are punctuation to the eye and syllables to the ear
# ("dash dash force"). Only the doubled form: a single hyphen has to survive
# for "bolt-pet" and "up-to-date", and requiring a word character after it
# keeps a "---" horizontal rule intact for _RULE to strip.
(re.compile(r"(?<!\w)--(?=\w)"), ""),
)
_MULTI_SPACE = re.compile(r"[ \t]+")
_MULTI_PUNCT = re.compile(r"(?:\s*\.){2,}")
@@ -97,38 +118,18 @@ def _bullets_to_sentences(text: str) -> str:
return " ".join(line if line[-1] in ".!?:,;" else line + "." for line in lines)
# ElevenLabs v3 delivery tags — "[laughing]", "[whispering]", "[sighs]". They
# are instructions to the *dialogue* model (see dialogue.py), and only inside a
# dialoguectl scene. Left in an ordinary reply they reach eleven_flash_v2,
# which has no idea what they mean and simply reads the word: observed live
# 2026-07-31, the pet announcing "laughing That one came through clean".
#
# Matched narrowly — a bracketed adverb/gerund, or one of the noise words that
# aren't either — so real bracketed text ("[1]", "[see the docs]") survives.
_AUDIO_TAG_RE = re.compile(
r"\[\s*(?:"
r"[a-z]+(?:ly|ing)"
r"|laughs?|chuckles?|sighs?|exhales?|inhales?|gulps?|pause|beat"
r"|clears throat|shouts?|whispers?|cries|sings?|gasps?|snorts?"
r"|sarcastic|excited|curious|nervous|angry|sad|happy|deadpan|flat"
r")\s*\]",
re.IGNORECASE,
)
def strip_audio_tags(text: str) -> str:
"""Remove v3 delivery tags from text destined for the ordinary voice."""
return _AUDIO_TAG_RE.sub(" ", str(text or ""))
def for_speech(text: str) -> str:
"""Plain prose for the TTS engine: no markdown, no emoji, no bare URLs,
no stray symbols that would be read out character by character."""
text = strip_audio_tags((text or "").strip())
text = (text or "").strip()
if not text:
return ""
for symbol, spoken in _PRE_SPOKEN_SYMBOLS.items():
text = text.replace(symbol, spoken)
# Before the markdown pass, so "#42" still has its "#" to word and a real
# "## Heading" (no digit after the hashes) is left for _HEADING to strip.
for pattern, spoken in _SPOKEN_ABBREVIATIONS:
text = pattern.sub(spoken, text)
text = _strip_markdown(text, keep_emoji=False)
text = _URL.sub(" link ", text)
text = _TABLE_PIPE.sub(", ", text)
@@ -143,28 +144,27 @@ def for_speech(text: str) -> str:
def is_question(text: str) -> bool:
"""True if the reply *ends* by asking the user something — the cue for
the pet to keep listening instead of making you say the wake word again.
"""True if the reply asks the user anything — the cue for the pet to keep
listening instead of making you say the wake word again.
Deliberately only looks at the end. A reply that asks something in
passing ("What time is it? It's 7:15.") isn't waiting on an answer,
whereas one that finishes on a question mark is. The test runs on the
spoken form, so a '?' that only exists inside a stripped code block or a
URL doesn't count, and trailing decoration (emoji, quotes, brackets) is
peeled off first so "Ready to go? 🚀" still reads as a question."""
spoken = for_speech(text)
while spoken and not (spoken[-1].isalnum() or spoken[-1] == "?"):
spoken = spoken[:-1]
return spoken.endswith("?")
Anywhere in the reply counts, not only the end. An earlier version required
a *trailing* '?' on the theory that "What time is it? It's 7:15." isn't
waiting on an answer, and that's true of that sentence but wrong far more
often: Bolt routinely asks first and then keeps talking ("Want me to fix
it? I'd start with the config."), and refusing to listen there is the case
that actually costs you a wake word. The cheap failure is the other
direction an unwanted extra listen ends itself on `VAD_GRACE_SECONDS` of
silence, and `FOLLOW_UP_MAX_TURNS` caps the chain.
The test runs on the *spoken* form, so a '?' that only exists inside a
stripped code block, a URL, or a markdown link target doesn't count."""
return "?" in for_speech(text)
def for_display(text: str) -> str:
"""What the speech bubble shows: markdown syntax removed (the bubble
can't render it) but emoji and layout-ish punctuation left alone.
Delivery tags go too the voice no longer says them, so showing them
would caption a laugh nobody heard."""
text = strip_audio_tags((text or "").strip())
can't render it) but emoji and layout-ish punctuation left alone."""
text = (text or "").strip()
if not text:
return ""
text = _strip_markdown(text, keep_emoji=True)
-2
View File
@@ -41,8 +41,6 @@ def run() -> int:
thread.started.connect(controller.run)
controller.state_changed.connect(lambda value: window.set_state(PetState(value)))
controller.said.connect(window.say)
# Lip-sync: the loudness of the audio actually going to the speakers.
controller.mouth.connect(window.set_mouth)
controller.log.connect(_log)
controller.action.connect(window.apply_action) # petctl move/emote/say/...
controller.finished.connect(thread.quit)
+2 -59
View File
@@ -19,7 +19,7 @@ from PySide6.QtGui import (
)
from PySide6.QtWidgets import QApplication, QWidget
from .. import config, window_state
from .. import config
from ..monitors import Monitor
from ..state import PetState
from .sprite import WALK, SpriteSet
@@ -36,11 +36,6 @@ _NAP_OPACITY = 0.35
# and the feet skate whenever PET_WANDER_SPEED doesn't happen to match the fps.
# Eight frames at 13px is a ~104px stride cycle, a bit under the pet's width.
_WALK_PIXELS_PER_FRAME = 13.0
# How long a loudness level stays believable. The audio thread sends one per
# ~30ms while a clip plays; if they stop arriving (offline TTS has no envelope,
# or playback died) the mouth must not stay frozen mid-syllable, so after this
# long the ordinary looping animation takes back over.
_MOUTH_STALE_SECONDS = 0.35
def emote_transform(emote: str, progress: float) -> tuple[float, float, float, float]:
@@ -218,9 +213,6 @@ class PetWindow(QWidget):
self._commanded_move = False # a petctl move — happens even mid-conversation
self._next_wander_at = 0.0
self._bob_offset = 0
# Lip-sync: loudness of what is playing right now, and when it arrived.
self._mouth_level: Optional[float] = None
self._mouth_at = 0.0
self._bob_phase = 0.0
self._walking = False
self._facing = 1 # +1 right, -1 left; the walk art is drawn facing right
@@ -257,19 +249,6 @@ class PetWindow(QWidget):
y = int(config.PET_START_Y) if config.PET_START_Y else None
except ValueError:
x = y = None
if x is None and y is None and config.PET_REMEMBER_POSITION:
remembered = window_state.load()
# Only if it still lands on a screen that exists — the usual reason
# a saved position is stale is that the monitor it was on has been
# unplugged, and restoring it faithfully would hide the pet.
if remembered is not None:
rectangles = [
(s.availableGeometry().left(), s.availableGeometry().top(),
s.availableGeometry().right(), s.availableGeometry().bottom())
for s in QApplication.screens()
]
if window_state.is_visible_on(*remembered, self.width(), rectangles):
x, y = remembered
if geo is not None:
x = geo.right() - self.width() - 40 if x is None else x
y = geo.bottom() - self.height() - 60 if y is None else y
@@ -709,34 +688,6 @@ class PetWindow(QWidget):
# ── animation ────────────────────────────────────────────────────────
def set_mouth(self, level: float) -> None:
"""How loud the pet is *right now* (0..1), straight off the PCM going
to the speakers (audio/tts.level_of).
The talking frames are ordered by mouth openness, so this indexes them
directly: the mouth moves with the actual waveform instead of flapping
on a timer, which is the difference between a talking sprite and a
dubbed one."""
self._mouth_level = max(0.0, min(1.0, float(level)))
self._mouth_at = time.monotonic()
if self._current_state == PetState.TALKING:
self.update()
def _mouth_frame(self, animation) -> Optional[QPixmap]:
"""The frame matching the current loudness, or None to use the timer.
Falls back the moment the levels go stale offline TTS has no
envelope, and a mouth frozen mid-syllable is worse than a timed loop."""
if self._mouth_level is None or animation is None:
return None
if time.monotonic() - self._mouth_at > _MOUTH_STALE_SECONDS:
return None
frames = animation.frames
if len(frames) < 2:
return None
index = int(round(self._mouth_level * (len(frames) - 1)))
return frames[max(0, min(len(frames) - 1, index))]
def _advance_frame(self) -> None:
# While walking the cycle is stepped by _advance_walk from distance
# travelled; letting this timer also advance it would double-step it
@@ -751,12 +702,7 @@ class PetWindow(QWidget):
painter.setRenderHint(QPainter.Antialiasing)
painter.setRenderHint(QPainter.SmoothPixmapTransform)
key = self._animation_key()
animation = self.sprites.get(key)
pixmap: Optional[QPixmap] = None
if key == PetState.TALKING:
pixmap = self._mouth_frame(animation)
if pixmap is None:
pixmap = animation.current()
pixmap: Optional[QPixmap] = self.sprites.get(key).current()
if key == WALK:
pixmap = self._oriented(pixmap)
if pixmap is None:
@@ -813,9 +759,6 @@ class PetWindow(QWidget):
was_click = not self._dragged
self._drag_offset = None
self._press_pos = None
if not was_click and config.PET_REMEMBER_POSITION:
position = self.geometry().topLeft()
window_state.save(position.x(), position.y())
if was_click:
self.talk_requested.emit()
else:
-75
View File
@@ -1,75 +0,0 @@
"""Where the pet was left, so it starts there next time.
Listed in the README as a known limitation: drag it somewhere deliberate, and
the next launch puts it back in the bottom-right corner. For something that
lives on your desktop all day that is a small daily annoyance, and it is a
config write on drag-end.
Lives in the cache dir rather than the repo, next to the restart context: it
is per-machine state about this install, not something that belongs in a git
diff. Every function swallows its own errors a corrupt or unwritable state
file must never stop the pet from starting, it just means the default corner.
Positions are validated against the *current* screen layout on load, because
the common case for a stale position is exactly the case where it is
dangerous: the pet was last on a monitor that is now unplugged, and restoring
it faithfully would put it somewhere you cannot see or reach.
"""
from __future__ import annotations
import json
import logging
import os
import tempfile
from pathlib import Path
from typing import Optional
logger = logging.getLogger("bolt_pet.window_state")
DEFAULT_PATH = Path.home() / ".cache" / "bolt-pet" / "window.json"
def load(path: Optional[Path] = None) -> Optional[tuple[int, int]]:
"""The saved position, or None if there isn't a usable one."""
try:
data = json.loads(Path(path or DEFAULT_PATH).read_text(encoding="utf-8"))
return int(data["x"]), int(data["y"])
except (FileNotFoundError, json.JSONDecodeError, KeyError, TypeError, ValueError, OSError):
return None
def save(x: int, y: int, path: Optional[Path] = None) -> None:
"""Remember where it is now. Atomic, so a crash mid-write can't leave a
half-file that makes the next start fall back to the corner."""
target = Path(path or DEFAULT_PATH)
try:
target.parent.mkdir(parents=True, exist_ok=True)
descriptor, temp_path = tempfile.mkstemp(dir=target.parent, prefix=".window_", suffix=".tmp")
try:
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
json.dump({"x": int(x), "y": int(y)}, handle)
os.replace(temp_path, target)
except BaseException:
try:
os.unlink(temp_path)
except OSError:
pass
raise
except Exception:
logger.debug("Could not save the window position", exc_info=True)
def is_visible_on(x: int, y: int, size: int, rectangles) -> bool:
"""Whether that position still lands on a screen that exists.
*rectangles* are (left, top, right, bottom) tuples the caller's job,
because this module has no business importing Qt. Requires a real overlap
rather than a touching edge, so a pet saved flush against the boundary of a
monitor that has since been unplugged is not counted as reachable."""
for left, top, right, bottom in rectangles:
overlap_x = min(x + size, right) - max(x, left)
overlap_y = min(y + size, bottom) - max(y, top)
if overlap_x > size * 0.25 and overlap_y > size * 0.25:
return True
return False
-3
View File
@@ -8,9 +8,6 @@ numpy>=1.24
# HTTP client to the Bolt desk API
requests>=2.31
# Streaming speech-to-text (audio/stt_stream.py). Optional in practice: without
# it the pet falls back to uploading the finished clip, exactly as before.
websocket-client>=1.7
# Wake-word detection (local, offline after first run) — runs the
# custom-trained thunderbolt.onnx model shipped in this repo, same runtime
+20 -48
View File
@@ -663,43 +663,31 @@ def render_frame(p) -> Image.Image:
def frames_for(state: str) -> list[dict]:
if state == "idle":
# Eight frames, all distinct. The old version drove breathing on
# sin(2*pi*t) and the tail on sin(4*pi*t), which both cross zero at
# i=0 and i=4 — so frame 4 was byte-identical to frame 0 and the loop
# was really four frames stored twice.
out = []
for i in range(8):
t = i / 8
br = math.sin(t * 2 * math.pi + math.pi / 7)
br = math.sin(t * 2 * math.pi)
out.append(
default_pose(
breathe=br,
head_dy=-0.006 * br,
# Three-halves harmonic: never in phase with the breath, so
# no two frames of the cycle can coincide.
tail=math.sin(t * 3 * math.pi + 0.6),
ear_twitch=0.30 if i == 3 else 0.0, # a flick, once a loop
tail=math.sin(t * 4 * math.pi),
blink=1.0 if i == 6 else 0.0,
)
)
return out
if state == "listening":
# Six frames of *orienting*, not idling: ears up, head turning toward
# whoever is talking, then settling. The old four had frames 0 and 2
# differing by 0.09 mean pixels — a two-pose animation wearing four.
out = []
for i in range(6):
t = i / 6
lean = math.sin(t * 2 * math.pi + math.pi / 5)
for i in range(4):
t = i / 4
out.append(
default_pose(
ear=1.0,
ear_twitch=0.45 * math.sin(t * 4 * math.pi),
tilt=-9 + 4.0 * lean,
look=(0.010 * lean, -0.004),
ear_twitch=0.35 * math.sin(t * 2 * math.pi),
tilt=-7 + 2.0 * math.sin(t * 2 * math.pi),
brow=1.0,
tail=0.7 * math.sin(t * 2 * math.pi + 1.1),
head_dy=-0.010 - 0.004 * lean,
tail=0.5 * math.sin(t * 2 * math.pi),
head_dy=-0.008,
tag_glow=True,
extras="listen",
phase=i,
@@ -707,51 +695,35 @@ def frames_for(state: str) -> list[dict]:
)
return out
if state == "thinking":
# The old six moved by a mean of ~1.3 pixels — effectively a still
# image. Thinking should *look* like thinking: the head tilts, the eyes
# travel as if following a thought, and one ear rotates independently.
out = []
for i in range(6):
t = i / 6
sway = math.sin(t * 2 * math.pi)
out.append(
default_pose(
ear=0.25 + 0.35 * abs(sway),
ear_twitch=0.5 * math.cos(t * 2 * math.pi),
tilt=4.0 + 7.0 * sway,
# Eyes wander a small circle: the cheapest possible read of
# "working something out" and the thing most obviously
# missing before.
look=(0.026 * math.cos(t * 2 * math.pi),
-0.020 + 0.014 * math.sin(t * 2 * math.pi)),
brow=0.5 + 0.4 * abs(sway),
breathe=0.5 * math.sin(t * 2 * math.pi + 0.9),
head_dy=-0.008 * sway,
tail=0.35 * math.sin(t * 3 * math.pi),
blink=1.0 if i == 4 else 0.0,
ear=0.25,
tilt=6.0,
look=(0.022, -0.026),
brow=0.5,
breathe=0.4 * math.sin(t * 2 * math.pi),
tail=0.2 * math.sin(t * 2 * math.pi),
extras="think",
phase=i // 2,
)
)
return out
if state == "talking":
# Ordered by mouth openness — closed at frame 0, widest at the last —
# because the window indexes these by the loudness of the audio that is
# actually playing (see audio/tts.level_of and PetWindow.set_mouth).
# A time-ordered loop cannot be indexed that way, and a mouth that
# flaps on a timer is what makes a talking sprite look dubbed.
out = []
count = 6
for i in range(count):
open_ = i / (count - 1)
for i in range(4):
t = i / 4
open_ = (math.sin(t * 2 * math.pi) + 1) / 2
out.append(
default_pose(
mouth=0.06 + 0.94 * open_,
mouth=0.25 + 0.75 * open_,
ear=0.6,
head_dy=-0.010 * open_,
breathe=open_,
tail=0.45 * math.sin(i * 0.9),
brow=0.35 * open_,
tail=math.sin(t * 2 * math.pi + 1.0),
brow=0.35,
)
)
return out
+5 -13
View File
@@ -22,14 +22,6 @@ def no_screen_probes(monkeypatch):
monkeypatch.setattr(controller_mod.screen_context, "is_fullscreen_active", lambda: False)
@pytest.fixture(autouse=True)
def _no_streaming(monkeypatch):
"""Most tests drive the plain request/response path; leaving streaming on
would have them attempt a real HTTP call, fail, and fall back passing,
slowly, for the wrong reason. The streaming path has its own tests."""
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", False)
@pytest.fixture
def ctrl():
return controller_mod.PetController()
@@ -49,10 +41,10 @@ def test_full_turn_happy_path(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's the weather")
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None, on_say=None: controller_mod.server_client.Reply("sunny and 72"))
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: controller_mod.server_client.Reply("sunny and 72"))
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: (spoken.append(text), True)[1])
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
ctrl._handle_conversation_turn()
@@ -67,7 +59,7 @@ def test_turn_with_nothing_heard_returns_to_idle_without_calling_server(monkeypa
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: None)
called = {"n": 0}
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: called.__setitem__("n", called["n"] + 1))
lambda text, on_command=None: called.__setitem__("n", called["n"] + 1))
ctrl._handle_conversation_turn()
@@ -105,7 +97,7 @@ def test_turn_with_server_error_flashes_error_then_idle(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "hello")
def boom(text, on_command=None, on_say=None):
def boom(text, on_command=None):
raise controller_mod.server_client.ServerError("server is down")
monkeypatch.setattr(controller_mod.server_client, "converse", boom)
@@ -158,7 +150,7 @@ def test_heartbeat_speaks_a_pending_announcement_when_idle(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: "don't forget your 3pm")
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: (spoken.append(text), True)[1])
lambda text, on_error=None, should_stop=None, voice_id=None: (spoken.append(text), True)[1])
said = _capture(ctrl.said)
ctrl._maybe_heartbeat()
+222 -260
View File
@@ -29,14 +29,6 @@ def no_screen_probes(monkeypatch):
monkeypatch.setattr(controller_mod.screen_context, "context_for", lambda text: text)
@pytest.fixture(autouse=True)
def _no_streaming(monkeypatch):
"""Most tests drive the plain request/response path; leaving streaming on
would have them attempt a real HTTP call, fail, and fall back passing,
slowly, for the wrong reason. The streaming path has its own tests."""
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", False)
@pytest.fixture
def ctrl():
return controller_mod.PetController()
@@ -168,7 +160,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch
ctrl._barge_in = detector
logs = _capture(ctrl.log)
def interrupted_playback(text, on_error=None, should_stop=None, voice_id=None, on_level=None):
def interrupted_playback(text, on_error=None, should_stop=None, voice_id=None):
# What really happens: frames get scored during playback, then one
# clears the threshold and playback aborts.
detector._frames, detector._peak, detector._last = 7, 0.81, 0.81
@@ -188,7 +180,7 @@ def test_the_interrupt_log_reports_what_fired_not_the_reset_counters(monkeypatch
def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
logs = _capture(ctrl.log)
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: False) # interrupted
lambda text, on_error=None, should_stop=None, voice_id=None: False) # interrupted
ctrl._speak("a very long explanation")
@@ -198,7 +190,7 @@ def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
def test_uninterrupted_playback_does_not_queue_a_turn(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
ctrl._speak("short answer")
assert not ctrl._talk_now.is_set()
@@ -210,7 +202,7 @@ def spoke(monkeypatch):
"""Playback that always completes, so only the follow-up rule decides
whether another turn is queued."""
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
def test_a_reply_ending_in_a_question_keeps_listening(spoke, ctrl):
@@ -229,10 +221,13 @@ def test_a_statement_does_not_keep_listening(spoke, ctrl):
assert not ctrl._pending_follow_up
def test_a_question_in_passing_does_not_count(spoke, ctrl):
"""Only a reply that *ends* on a question is waiting for an answer."""
ctrl._speak("What time is it? It's 7:15 AM.")
assert not ctrl._talk_now.is_set()
def test_a_question_anywhere_in_the_reply_keeps_listening(spoke, ctrl):
"""Bolt often asks and then keeps talking ("Want me to fix it? I'd start
with the config."), so the question mark doesn't have to be last. An
unwanted extra listen ends itself on VAD_GRACE_SECONDS of silence; a missed
one costs you a wake word, which is the more expensive mistake."""
ctrl._speak("Want me to restart it? It's been up for 40 days.")
assert ctrl._talk_now.is_set()
def test_follow_ups_stop_at_the_cap(spoke, monkeypatch, ctrl):
@@ -298,7 +293,7 @@ def test_follow_up_can_be_turned_off(spoke, monkeypatch, ctrl):
def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: False)
lambda text, on_error=None, should_stop=None, voice_id=None: False)
ctrl._follow_ups = 3
ctrl._speak("a very long explanation")
@@ -309,7 +304,7 @@ def test_being_interrupted_restarts_the_chain(monkeypatch, ctrl):
def test_speech_is_recorded_in_the_history(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
ctrl._speak("**bold** reply")
assert ctrl.history.last().text == "**bold** reply" # raw, for copy/paste
@@ -323,10 +318,10 @@ def test_the_active_window_rides_along_with_the_utterance(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.screen_context, "context_for",
lambda text: f"{text}\n\n[on screen right now: app.py]")
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: sent.append(text) or Reply("that's a KeyError"))
lambda text, on_command=None: sent.append(text) or Reply("that's a KeyError"))
ctrl._handle_conversation_turn()
@@ -379,10 +374,10 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "you awake?")
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: Reply("always"))
lambda text, on_command=None: Reply("always"))
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: spoken.append(text) or True)
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
ctrl.set_napping(True)
ctrl._handle_conversation_turn()
@@ -390,6 +385,148 @@ def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
assert spoken == ["always"]
# ── local intents ───────────────────────────────────────────────────────────
@pytest.fixture
def heard(monkeypatch):
"""A turn where you said something, with the server and TTS recorded.
Returns (utterance_setter, sent, spoken)."""
said = {"text": ""}
sent, spoken = [], []
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said["text"])
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or Reply("from the server"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None:
spoken.append(text) or True)
return said, sent, spoken
def test_a_local_intent_never_reaches_the_server(heard, ctrl):
said, sent, spoken = heard
said["text"] = "come here"
actions = _capture(ctrl.action)
ctrl._handle_conversation_turn()
assert sent == [] # no round trip at all
assert actions == [{"action": "move", "anchor": "cursor"}]
assert spoken == [] # walking over is the reply
assert ctrl._state.state == PetState.IDLE
def test_stop_is_answered_with_silence(heard, ctrl):
said, sent, spoken = heard
said["text"] = "be quiet"
ctrl._handle_conversation_turn()
assert (sent, spoken) == ([], [])
def test_a_request_that_merely_starts_with_an_intent_word_goes_to_the_server(heard, ctrl):
said, sent, spoken = heard
said["text"] = "stop the docker container"
ctrl._handle_conversation_turn()
assert sent and "stop the docker container" in sent[0]
assert spoken == ["from the server"]
def test_local_intents_are_skipped_while_answering_a_question(heard, ctrl):
"""Bolt asked something; "never mind" is an answer to him, not a body
command. Swallowing it locally would leave the server holding a question it
never got a reply to."""
said, sent, spoken = heard
said["text"] = "never mind"
ctrl._pending_follow_up = True
ctrl._handle_conversation_turn()
assert sent and "never mind" in sent[0]
def test_local_intents_can_be_turned_off(monkeypatch, heard, ctrl):
monkeypatch.setattr(controller_mod.config, "LOCAL_INTENTS", False)
said, sent, spoken = heard
said["text"] = "come here"
ctrl._handle_conversation_turn()
assert sent and "come here" in sent[0]
def test_say_that_again_replays_the_last_line_without_duplicating_history(heard, ctrl):
said, sent, spoken = heard
ctrl.history.add(controller_mod.history_mod.PET, "it's 7:15 AM", 0.0)
said["text"] = "what did you say?"
ctrl._handle_conversation_turn()
assert spoken == ["it's 7:15 AM"]
assert sent == []
pet_lines = [e.text for e in ctrl.history.entries()
if e.role == controller_mod.history_mod.PET]
assert pet_lines == ["it's 7:15 AM"] # replayed, not re-recorded
def test_repeat_with_nothing_to_repeat_says_so(heard, ctrl):
said, sent, spoken = heard
said["text"] = "say that again"
ctrl._handle_conversation_turn()
assert spoken == ["I haven't said anything yet."]
def test_going_back_to_the_normal_voice_needs_no_server_prompt_support(heard, ctrl):
"""The server can only offer `petctl voice reset` if its prompt happens to
advertise the verb; recognising the phrase here works regardless."""
said, sent, spoken = heard
ctrl._voice_id, ctrl._voice_name = "voice-123", "Brian"
changed = _capture(ctrl.voice_changed)
said["text"] = "go back to your normal voice"
ctrl._handle_conversation_turn()
assert ctrl._voice_id == ""
assert changed == [""]
assert spoken == ["Back to my own voice."]
assert sent == []
def test_go_to_sleep_overrides_the_quiet_hours_schedule(heard, ctrl):
said, sent, spoken = heard
said["text"] = "go to sleep"
ctrl._handle_conversation_turn()
assert ctrl._napping is True
assert ctrl._nap_forced is True # not undone by the next schedule check
assert spoken == ["Night."]
# ── failure containment ─────────────────────────────────────────────────────
def test_one_bad_turn_does_not_end_the_session(monkeypatch, ctrl):
"""A turn raising something unforeseen used to unwind _loop and kill the
thread the pet would go deaf until it was restarted by hand."""
logs = _capture(ctrl.log)
def explode():
raise RuntimeError("numpy said no")
assert ctrl._guarded(explode, "conversation turn") is False
assert ctrl._state.state == PetState.IDLE
assert any("Recovered from a conversation turn failure" in m for m in logs)
def test_a_command_handler_crash_is_reported_up_the_relay(monkeypatch, ctrl):
"""The server is blocked on /desk/tool_result while this runs. Raising would
leave it waiting out its own timeout on a turn that can never finish."""
monkeypatch.setattr(controller_mod.pet_actions, "parse",
lambda command: (_ for _ in ()).throw(KeyError("boom")))
output = ctrl._handle_command("petctl move top-left")
assert output.startswith("[error]") and "boom" in output
# ── notification bridge ─────────────────────────────────────────────────────
def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
@@ -397,9 +534,9 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
sent, spoken = [], []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: sent.append(text) or Reply("your build is green"))
lambda text, on_command=None: sent.append(text) or Reply("your build is green"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: spoken.append(text) or True)
lambda text, on_error=None, should_stop=None, voice_id=None: spoken.append(text) or True)
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
ctrl._drain_notifications()
@@ -412,14 +549,61 @@ def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
def test_filtered_out_notifications_are_never_queued(ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("deploy", 0)
ctrl._queue_notification(Notification(app="Chat", summary="lunch?", body=""))
assert ctrl._pending_notifications == []
assert not ctrl._pending_notifications
def test_the_notification_queue_is_bounded(monkeypatch, ctrl):
"""An overnight nap can't grow the queue without limit — the drain only runs
from the heartbeat, and the heartbeat doesn't run while napping."""
monkeypatch.setattr(controller_mod.config, "NOTIFICATION_QUEUE_LIMIT", 3)
ctrl = controller_mod.PetController()
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
for index in range(10):
ctrl._queue_notification(Notification(app="CI", summary=f"build {index}", body=""))
queued = [notification.summary for _stamp, notification in ctrl._pending_notifications]
assert queued == ["build 7", "build 8", "build 9"] # oldest dropped
def test_stale_notifications_are_dropped_instead_of_read_out(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "NOTIFICATION_MAX_AGE_SECONDS", 900)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None: sent.append(text) or Reply(""))
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
ctrl._queue_notification(Notification(app="CI", summary="fresh", body=""))
# Backdate it past the age limit, as an overnight backlog would be.
stamp, notification = ctrl._pending_notifications.pop()
ctrl._pending_notifications.append((stamp - 4000, notification))
ctrl._drain_notifications()
assert sent == []
def test_a_nap_starting_mid_drain_keeps_the_rest_queued(monkeypatch, ctrl):
"""The old code swapped the queue out and returned, losing the remainder."""
def converse(text, on_command=None):
ctrl._napping = True # e.g. quiet hours began, or a fullscreen app opened
return Reply("")
monkeypatch.setattr(controller_mod.server_client, "converse", converse)
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
for index in range(3):
ctrl._queue_notification(Notification(app="CI", summary=f"build {index}", body=""))
ctrl._drain_notifications()
remaining = [notification.summary for _stamp, notification in ctrl._pending_notifications]
assert remaining == ["build 1", "build 2"]
def test_notifications_are_not_forwarded_while_napping(monkeypatch, ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
called = {"n": 0}
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: called.__setitem__("n", called["n"] + 1))
lambda text, on_command=None: called.__setitem__("n", called["n"] + 1))
ctrl.set_napping(True)
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
@@ -515,9 +699,9 @@ def test_check_deliveries_runs_after_a_conversation_turn(monkeypatch, ctrl, tmp_
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "send me that file")
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: Reply("it's on the way"))
lambda text, on_command=None: Reply("it's on the way"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
lambda text, on_error=None, should_stop=None, voice_id=None: True)
monkeypatch.setattr(controller_mod.config, "DELIVERED_FILES_DIR", tmp_path)
monkeypatch.setattr(controller_mod.server_client, "list_outbox_files",
lambda: [{"id": "abc", "name": "notes.txt", "size": 2}])
@@ -539,10 +723,10 @@ def _voice_turn(monkeypatch, ctrl, reply, said="talk like a pirate"):
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: said)
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: reply)
lambda text, on_command=None: reply)
monkeypatch.setattr(
controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None:
lambda text, on_error=None, should_stop=None, voice_id=None:
voices.append(voice_id) or True,
)
ctrl._handle_conversation_turn()
@@ -633,7 +817,7 @@ def test_dialoguectl_never_reaches_the_shell(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm",
lambda pcm, rate, should_stop=None, on_level=None: True)
lambda pcm, rate, should_stop=None: True)
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "[cheerfully] hi"}))
@@ -646,7 +830,7 @@ def test_a_scene_shows_in_the_bubble_with_the_delivery_tags_stripped(monkeypatch
monkeypatch.setattr(controller_mod.config, "DIALOGUE_VOICES", "narrator:9BWtsMINqrJLrRacOk9x")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None, on_level=None: True)
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
said = _capture(ctrl.said)
ctrl._handle_command(_dialogue_command(
@@ -664,7 +848,7 @@ def test_a_mid_turn_scene_returns_to_thinking_not_idle(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "ELEVENLABS_VOICE_ID", "aaorr6ZHIL88gEexu7dC")
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None, on_level=None: True)
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
ctrl._state.transition(PetState.LISTENING)
ctrl._state.transition(PetState.THINKING)
states = _capture(ctrl.state_changed)
@@ -684,7 +868,7 @@ def test_the_scene_uses_a_voice_the_server_picked_with_speak_as(monkeypatch, ctr
return np.zeros(4, dtype=np.int16), 24000
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue", capture)
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None, on_level=None: True)
monkeypatch.setattr(controller_mod.tts, "play_pcm", lambda pcm, rate, should_stop=None: True)
ctrl._apply_voice(controller_mod.server_client.Reply("ok", "PICKEDvoice123456789", "Terence"))
ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
@@ -726,7 +910,7 @@ def test_talking_over_a_scene_is_reported_up_the_relay(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "synthesize_dialogue",
lambda inputs, model_id=None, stability=None: (np.zeros(4, dtype=np.int16), 24000))
monkeypatch.setattr(controller_mod.tts, "play_pcm",
lambda pcm, rate, should_stop=None, on_level=None: False) # barge-in
lambda pcm, rate, should_stop=None: False) # barge-in
output = ctrl._handle_command(_dialogue_command({"voice": "self", "text": "hi"}))
@@ -796,9 +980,9 @@ def test_coming_back_up_reports_to_the_server_and_speaks_the_reply(monkeypatch,
path=state, now=1000.0)
sent, spoken = [], []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: sent.append(text) or Reply("Good, it's up."))
lambda text, on_command=None: sent.append(text) or Reply("Good, it's up."))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None:
lambda text, on_error=None, should_stop=None, voice_id=None:
spoken.append(text) or True)
ctrl._report_self_restart()
@@ -813,230 +997,8 @@ def test_an_ordinary_start_reports_nothing(monkeypatch, ctrl, tmp_path):
monkeypatch.setattr(controller_mod.self_restart, "DEFAULT_STATE_PATH", tmp_path / "none.json")
called = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: called.append(text))
lambda text, on_command=None: called.append(text))
ctrl._report_self_restart()
assert called == []
# ── holding line: "give me a sec" while a tool runs ────────────────────────
# The server used to discard whatever the model wrote alongside a tool call,
# so the whole round trip was silence — and the model, with no evidence its
# sentence landed, said it again in the final reply.
def test_a_holding_line_is_spoken_before_the_tool_runs(monkeypatch, ctrl):
order = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None:
order.append(("spoke", text)) or True)
monkeypatch.setattr(controller_mod.server_client, "run_local_command",
lambda cmd: order.append(("ran", cmd)) or "[exit 0]")
ctrl._speak_holding("Give me a sec.")
ctrl._handle_command("df -h /")
assert order == [("spoke", "Give me a sec."), ("ran", "df -h /")]
def test_the_holding_line_shows_in_the_bubble_but_not_the_transcript(monkeypatch, ctrl):
"""It's filler. The transcript should keep the actual answer."""
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
said = _capture(ctrl.said)
ctrl._speak_holding("I'll check on that — one moment.")
assert said == ["I'll check on that — one moment."]
assert ctrl.history.entries() == []
def test_a_holding_line_returns_to_waiting_not_idle(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
ctrl._state.transition(PetState.LISTENING)
ctrl._state.transition(PetState.THINKING)
states = _capture(ctrl.state_changed)
ctrl._speak_holding("one sec")
assert states == ["talking", "thinking"]
def test_the_relay_speaks_whatever_the_server_attaches(monkeypatch, ctrl):
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None:
spoken.append(text) or True)
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's in that folder?")
def fake_converse(text, on_command=None, on_say=None):
on_say("Let me check that for you.") # what the server now sends
on_command("filectl {\"op\": \"list\", \"path\": \"/tmp\"}")
return Reply("It's got three files in it.")
monkeypatch.setattr(controller_mod.server_client, "converse", fake_converse)
ctrl._handle_conversation_turn()
assert spoken == ["Let me check that for you.", "It's got three files in it."]
# ── streamed replies ───────────────────────────────────────────────────────
# Without streaming the pet waits out the entire model call before saying a
# word. With it, the wait is time-to-first-sentence.
def test_each_sentence_is_spoken_as_it_arrives(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", True)
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None:
spoken.append(text) or True)
def fake_stream(text, on_say, on_command=None, timeout=180.0):
on_say("The disk is fine.")
on_say("About sixty percent used.")
return controller_mod.server_client.Reply(
"The disk is fine. About sixty percent used.", spoken=True)
monkeypatch.setattr(controller_mod.server_client, "converse_stream", fake_stream)
monkeypatch.setattr(controller_mod.mic, "record_utterance",
lambda *a, **k: np.zeros(10, dtype=np.int16))
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "how's the disk?")
ctrl._handle_conversation_turn()
assert spoken == ["The disk is fine.", "About sixty percent used."]
# ...and the empty final reply is not spoken as an extra blank utterance.
assert ctrl.history.entries()[-1].text == "About sixty percent used."
def test_a_stream_that_fails_before_speaking_falls_back_silently(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", True)
spoken = []
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None:
spoken.append(text) or True)
def broken_stream(text, on_say, on_command=None, timeout=180.0):
raise controller_mod.server_client.ServerError("no streaming endpoint")
monkeypatch.setattr(controller_mod.server_client, "converse_stream", broken_stream)
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: Reply("Fell back fine."))
assert ctrl._ask_server("hello").text == "Fell back fine."
assert spoken == [] # nothing was said twice
def test_streaming_can_be_switched_off(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", False)
called = []
monkeypatch.setattr(controller_mod.server_client, "converse_stream",
lambda *a, **k: called.append(1))
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: Reply("plain path"))
assert ctrl._ask_server("hi").text == "plain path"
assert called == []
def test_talking_over_a_streamed_reply_still_interrupts(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", True)
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: False)
ctrl._speak_stream_chunk("A long explanation you cut short.")
assert ctrl._talk_now.is_set()
# ── notification latency (reported 2026-08-02: 5-10 minutes) ───────────────
# Draining was wired to the heartbeat's 60s interval, and a heartbeat that
# landed mid-conversation stamped its clock before noticing — so it burned the
# slot and waited another full interval. Several of those in a row is minutes.
def test_a_notification_goes_out_on_the_next_tick_not_the_next_heartbeat(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.config, "NOTIFICATION_MIN_INTERVAL_SECONDS", 0)
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None:
sent.append(text) or Reply("noted"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: None)
# A heartbeat has just run, so the next one is a full interval away.
ctrl._last_heartbeat = controller_mod.time.monotonic()
ctrl._queue_notification(Notification(app="Signal", summary="Harry", body="you there?"))
ctrl._maybe_heartbeat()
assert sent and "Harry" in sent[0]
def test_a_heartbeat_skipped_mid_conversation_does_not_burn_its_slot(monkeypatch, ctrl):
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: None)
ctrl._last_heartbeat = 0.0
ctrl._state.transition(PetState.LISTENING)
ctrl._state.transition(PetState.THINKING)
ctrl._maybe_heartbeat() # due, but the pet is busy
assert ctrl._last_heartbeat == 0.0, "the clock must not advance on a skipped tick"
ctrl._state.transition(PetState.IDLE)
ctrl._maybe_heartbeat() # free now — runs immediately
assert ctrl._last_heartbeat > 0.0
def test_notifications_wait_while_the_pet_is_mid_turn(monkeypatch, ctrl):
"""Speaking over the answer they're already getting would be worse than
waiting a couple of seconds."""
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None: sent.append(text))
ctrl._state.transition(PetState.LISTENING)
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
ctrl._maybe_heartbeat()
assert sent == []
assert len(ctrl._pending_notifications) == 1 # kept, not dropped
def test_a_second_message_inside_the_rate_limit_is_no_longer_lost(monkeypatch, ctrl):
"""The gate used to drop it. With NOTIFICATION_MIN_INTERVAL_SECONDS=60,
two texts a minute apart meant you heard about one of them."""
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 60)
sent = []
monkeypatch.setattr(controller_mod.server_client, "converse",
lambda text, on_command=None, on_say=None:
sent.append(text) or Reply("ok"))
monkeypatch.setattr(controller_mod.tts, "speak",
lambda text, on_error=None, should_stop=None, voice_id=None, on_level=None: True)
ctrl._queue_notification(Notification(app="Signal", summary="Harry", body="you there?"))
ctrl._queue_notification(Notification(app="Signal", summary="Harry", body="it's urgent"))
ctrl._drain_notifications()
assert len(sent) == 1 # one round trip...
assert "you there?" in sent[0] and "it's urgent" in sent[0] # ...both messages
assert "2 desktop notifications" in sent[0]
def test_the_filter_still_applies(ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("deploy", 0)
ctrl._queue_notification(Notification(app="Chat", summary="lunch?", body=""))
assert ctrl._pending_notifications == []
def test_a_notification_storm_cannot_become_an_unbounded_backlog(ctrl):
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
for index in range(40):
ctrl._queue_notification(Notification(app="Spam", summary=f"#{index}", body=""))
assert len(ctrl._pending_notifications) == controller_mod._MAX_PENDING_NOTIFICATIONS
assert "#39" in ctrl._pending_notifications[-1].as_text() # newest kept
+1 -12
View File
@@ -161,18 +161,7 @@ def test_the_relay_report_names_the_cast():
action = dialogue.parse('dialoguectl ' + _scene(
'{"voice": "self", "text": "one"}', '{"voice": "narrator", "text": "two"}',
))
assert dialogue.describe(action).startswith(
"[dialogue] played 2 lines in 2 voices: narrator, self")
def test_the_report_says_the_scene_was_already_heard():
"""Observed 2026-07-31: the scene played, then the final reply summarised
it, so the pet said the same thing twice with nothing in between. The
model cannot know the audio already happened unless it is told."""
action = dialogue.parse('dialoguectl {"lines": [{"text": "hello"}]}')
report = dialogue.describe(action)
assert "HEARD this already" in report
assert "Do not repeat" in report
assert dialogue.describe(action) == "[dialogue] played 2 lines in 2 voices: narrator, self"
# ── the HTTP request ────────────────────────────────────────────────────────
-115
View File
@@ -1,115 +0,0 @@
"""The preflight. Its one job is to never be the thing that's broken.
A doctor that raises on a broken install diagnoses the wrong patient, so the
tests that matter here are the ugly-input ones: no config at all, a check that
throws, a dependency missing. The individual diagnoses are simple enough to
read; that they *run* on a machine missing everything is the property worth
pinning down.
"""
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import config, doctor
from bolt_pet.doctor import FAIL, OK, WARN
def test_every_check_returns_a_verdict_on_a_bare_machine(monkeypatch):
"""Nothing configured, nothing installed — still a full report."""
for name in ("SERVER_URL", "API_KEY", "DEEPGRAM_API_KEY",
"ELEVENLABS_API_KEY", "ELEVENLABS_VOICE_ID"):
monkeypatch.setattr(config, name, "")
monkeypatch.setattr(doctor, "_module", lambda _n: False)
results = doctor.run()
assert len(results) == len(doctor.CHECKS)
assert all(c.status in (OK, WARN, FAIL) for c in results)
assert all(c.name and c.detail for c in results)
def test_a_check_that_raises_does_not_hide_the_others():
"""One broken probe must not cost you the other eleven diagnoses."""
def explode(*_args):
raise RuntimeError("boom")
original = doctor.CHECKS
doctor.CHECKS = (("mic", explode),) + original[:2]
try:
results = doctor.run()
finally:
doctor.CHECKS = original
assert len(results) == 3
assert results[0].status == FAIL
assert "boom" in results[0].detail
def test_missing_server_config_is_a_failure_not_a_warning(monkeypatch):
"""Without it the controller exits its thread at startup — the pet looks
alive and simply never answers. That is the worst failure mode there is."""
monkeypatch.setattr(config, "missing_config", lambda: ["BOLT_SERVER_URL", "DESK_API_KEY"])
check = doctor.check_config()
assert check.status == FAIL
assert "BOLT_SERVER_URL" in check.detail
assert check.fix
def test_a_configured_server_passes_without_being_contacted(monkeypatch):
"""The shallow run must not need the network — it's the first thing you
reach for when the network is what's wrong."""
monkeypatch.setattr(config, "missing_config", lambda: [])
monkeypatch.setattr(config, "SERVER_URL", "http://bolt.local:8000")
assert doctor.check_config().status == OK
assert doctor.check_server(deep=False).status == OK
def test_every_problem_comes_with_something_to_do_about_it(monkeypatch):
""""screen reading: warn" is useless on its own; "apt install tesseract-ocr"
is the entire point of the tool."""
for name in ("SERVER_URL", "API_KEY", "DEEPGRAM_API_KEY"):
monkeypatch.setattr(config, name, "")
monkeypatch.setattr(doctor, "_module", lambda _n: False)
for check in doctor.run():
if check.status == FAIL:
assert check.fix, f"{check.name} says what's wrong but not what to do"
def test_the_exit_code_is_nonzero_only_for_real_failures(monkeypatch, capsys):
monkeypatch.setattr(doctor, "run", lambda deep=False: [
doctor.Check("a", OK, "fine"), doctor.Check("b", WARN, "degraded")])
assert doctor.main([]) == 0
monkeypatch.setattr(doctor, "run", lambda deep=False: [doctor.Check("a", FAIL, "broken")])
assert doctor.main([]) == 1
assert "Fix those first" in capsys.readouterr().out
def test_deep_is_off_unless_asked(monkeypatch):
seen = []
monkeypatch.setattr(doctor, "run", lambda deep=False: seen.append(deep) or [])
doctor.main([])
doctor.main(["--deep"])
assert seen == [False, True]
def test_a_slow_silence_timeout_is_flagged(monkeypatch):
"""The setting most likely to make it feel sluggish, and the least obvious
it is pure dead air before anything at all starts happening."""
monkeypatch.setattr(config, "SILENCE_END_SEC", 2.0)
check = doctor.check_latency()
assert check.status == WARN
assert "2s" in check.detail or "2 " in check.detail
monkeypatch.setattr(config, "SILENCE_END_SEC", 0.9)
assert doctor.check_latency().status == OK
def test_a_check_line_renders_the_fix_only_when_there_is_a_problem():
assert "" not in doctor.Check("x", OK, "all good", fix="unused").line()
assert "→ do the thing" in doctor.Check("x", WARN, "hmm", fix="do the thing").line()
+112
View File
@@ -0,0 +1,112 @@
"""Local intent recognition — pure string logic, no hardware or display.
The interesting tests are the negative ones: this feature's whole risk is
swallowing something that was meant for the server.
"""
import sys
from pathlib import Path
import pytest
from bolt_pet import intents
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
@pytest.mark.parametrize("said, expected", [
("stop", "stop"),
("Stop!", "stop"),
("never mind", "stop"),
("be quiet", "stop"),
("go to sleep", "nap"),
("take a nap", "nap"),
("goodnight", "nap"),
("wake up", "wake"),
("come here", "come"),
("follow my cursor", "come"),
("get out of the way", "go_away"),
("hide", "go_away"),
("say that again", "repeat"),
("what did you say?", "repeat"),
("go for a walk", "wander_on"),
("stay put", "wander_off"),
("sit", "wander_off"),
("use your normal voice", "voice_reset"),
("go back to your normal voice", "voice_reset"),
("be yourself again", "voice_reset"),
])
def test_recognized_phrases(said, expected):
intent = intents.recognize(said)
assert intent is not None and intent.name == expected
@pytest.mark.parametrize("said", [
# Each of these starts with (or contains) an intent phrase, and every one is
# a real request. A substring match would eat all of them.
"stop the docker container",
"stop the deploy and tell me what broke",
"can you hide the window that's covering my terminal",
"come up with a name for this branch",
"repeat the last command but with sudo",
"what did you say the disk usage was on the server",
"move the config file to the backup directory",
"sit down and write me a haiku about kubernetes",
"what time is it",
"go to sleep mode on the server",
"",
" ",
# Pure filler leaves an empty string, which must not match anything.
"hey bolt",
"okay bolt please",
])
def test_real_requests_are_left_for_the_server(said):
assert intents.recognize(said) is None
def test_filler_is_stripped_from_both_ends():
assert intents.normalize("Hey Bolt, could you please just stop now?") == "stop"
assert intents.normalize("okay, come here buddy") == "come here"
def test_normalize_returns_empty_for_pure_filler():
assert intents.normalize("hey bolt") == ""
assert intents.normalize("...") == ""
def test_intents_carry_ui_actions_in_the_pet_actions_shape():
"""The action dicts go straight to PetWindow.apply_action, so they have to
match the vocabulary pet_actions.parse produces no new UI cases."""
assert intents.recognize("come here").action == {"action": "move", "anchor": "cursor"}
assert intents.recognize("stay put").action == {"action": "wander", "enabled": False}
assert intents.recognize("go to sleep").action == {"action": "nap", "enabled": True}
def test_stop_says_nothing():
"""Answering "okay!" when told to be quiet defeats the purpose."""
intent = intents.recognize("be quiet")
assert intent.speak == "" and intent.action is None
def test_a_phrase_claimed_by_two_intents_fails_at_import(monkeypatch):
"""Without this guard the phrase would silently bind to whichever intent was
declared last a table edit that looks fine and misbehaves on a mic."""
monkeypatch.setattr(intents, "_TABLE", (
(intents.Intent("stop"), ("enough",)),
(intents.Intent("nap"), ("enough",)),
))
with pytest.raises(ValueError, match="claimed by both"):
intents._build()
def test_a_phrase_of_pure_filler_fails_at_import(monkeypatch):
"""It would normalise to "" and then match any all-filler utterance."""
monkeypatch.setattr(intents, "_TABLE", ((intents.Intent("stop"), ("please bolt",)),))
with pytest.raises(ValueError, match="normalises to nothing"):
intents._build()
def test_every_table_phrase_round_trips():
for phrase, intent in intents._BY_PHRASE.items():
assert phrase, "a phrase normalised to nothing"
assert intents.recognize(phrase) is intent
-135
View File
@@ -1,135 +0,0 @@
"""Amplitude → mouth openness, from PCM samples to the drawn frame.
Two halves, tested separately because only one of them needs a display:
`tts.level_of`/`envelope` turn samples into a 0..1 loudness, and
`PetWindow._mouth_frame` turns that loudness into a frame index. They meet at
the `mouth` signal, which the pipeline smoke test covers end to end.
"""
import sys
import time
from pathlib import Path
import numpy as np
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.audio import tts
def frame(amplitude: int, samples: int = 480) -> np.ndarray:
return np.full(samples, amplitude, dtype=np.int16)
# ── loudness ────────────────────────────────────────────────────────────────
def test_silence_closes_the_mouth():
assert tts.level_of(np.zeros(480, dtype=np.int16)) == 0.0
def test_a_loud_frame_opens_it_fully():
assert tts.level_of(frame(30000)) == 1.0
def test_the_level_never_leaves_zero_to_one():
"""It indexes a frame list; out of range is an IndexError on the UI thread."""
for amplitude in (0, 1, 500, 6000, 20000, 32767):
assert 0.0 <= tts.level_of(frame(amplitude)) <= 1.0
def test_it_rises_with_amplitude():
quiet, middling, loud = (tts.level_of(frame(a)) for a in (800, 4000, 12000))
assert quiet < middling < loud
def test_quiet_speech_still_moves_the_mouth_visibly():
"""The sqrt curve is the whole point. Speech spends most of its time well
below peak, so a linear map leaves the mouth barely open for normal talking
and the pet looks like it's mumbling."""
assert tts.level_of(frame(1500)) > 0.15
def test_an_empty_frame_is_silence_not_a_crash():
assert tts.level_of(np.zeros(0, dtype=np.int16)) == 0.0
# ── the envelope of a whole clip ────────────────────────────────────────────
def test_an_envelope_has_one_value_per_frame_at_the_requested_rate():
one_second = np.zeros(16000, dtype=np.int16)
assert len(tts.envelope(one_second, 16000, fps=30)) == pytest.approx(30, abs=1)
def test_an_envelope_tracks_loud_and_quiet_stretches():
pcm = np.concatenate([np.zeros(8000, dtype=np.int16), frame(20000, 8000)])
levels = tts.envelope(pcm, 16000, fps=10)
assert max(levels[:4]) == 0.0 # the silent half
assert min(levels[-4:]) > 0.5 # the loud half
def test_an_empty_clip_has_an_empty_envelope():
assert tts.envelope(np.zeros(0, dtype=np.int16), 16000) == []
# ── the drawn frame ─────────────────────────────────────────────────────────
@pytest.fixture
def window():
"""Needs a QApplication; run with QT_QPA_PLATFORM=offscreen."""
from PySide6.QtWidgets import QApplication
from bolt_pet.ui.pet_window import PetWindow
_app = QApplication.instance() or QApplication(["test"])
win = PetWindow()
yield win
win.close()
class FakeAnimation:
"""Stands in for sprite.Animation — _mouth_frame only wants `.frames`."""
def __init__(self, frames):
self.frames = list(frames)
def test_the_talking_frame_follows_the_level(window):
"""The talking frames are an openness ramp — closed first, widest last —
specifically so loudness can index them directly."""
animation = FakeAnimation(["closed", "a", "b", "c", "d", "open"])
assert window._mouth_frame(animation) is None # no level set yet
window.set_mouth(0.0)
assert window._mouth_frame(animation) == "closed"
window.set_mouth(1.0)
assert window._mouth_frame(animation) == "open"
window.set_mouth(0.5)
assert window._mouth_frame(animation) not in ("closed", "open")
def test_a_stale_level_gives_the_animation_back(window):
"""If the audio thread stops sending levels — TTS died, the clip ended
without a final zero the mouth must not freeze half-open forever. After a
moment it falls back to the ordinary looping animation."""
animation = FakeAnimation(["a", "b", "c"])
window.set_mouth(0.9)
assert window._mouth_frame(animation) is not None
window._mouth_at = time.monotonic() - 5.0
assert window._mouth_frame(animation) is None
def test_a_single_frame_animation_falls_back_instead_of_indexing(window):
"""Art with one talking frame can't lip-sync; it must not try."""
window.set_mouth(1.0)
assert window._mouth_frame(FakeAnimation(["only"])) is None
assert window._mouth_frame(FakeAnimation([])) is None
def test_no_animation_at_all_is_handled(window):
window.set_mouth(0.5)
assert window._mouth_frame(None) is None
+48
View File
@@ -112,3 +112,51 @@ def test_pcm_to_wav_bytes_round_trips_via_wave_module():
assert wf.getframerate() == 16000
frames = wf.readframes(wf.getnframes())
assert np.frombuffer(frames, dtype=np.int16).tolist() == pcm.tolist()
# ── flushing buffered audio ─────────────────────────────────────────────────
class _BufferedStream:
"""A stream with a backlog, like PortAudio's ring buffer after the reader
was blocked on a network call for a while."""
def __init__(self, available):
self.read_available = available
self.reads = []
def read(self, frames):
self.reads.append(frames)
self.read_available = max(0, self.read_available - frames)
return np.zeros((frames, 1), dtype=np.int16), False
def test_flush_drops_exactly_what_was_buffered():
stream = _BufferedStream(4096)
assert mic.flush(stream) == 4096
assert stream.reads == [4096]
assert stream.read_available == 0
def test_flush_is_bounded_so_it_cannot_chase_a_live_stream():
"""A stream filling as fast as it drains must not spin forever."""
stream = _BufferedStream(10 ** 9)
dropped = mic.flush(stream, max_seconds=1.0, sample_rate=16000)
assert dropped == 16000
def test_flush_is_a_noop_on_an_empty_or_fake_stream():
stream = _BufferedStream(0)
assert mic.flush(stream) == 0
assert stream.reads == []
assert mic.flush(_ScriptedStream([])) == 0 # no read_available at all
assert mic.flush(None) == 0
def test_flush_swallows_a_device_error():
class _Broken:
read_available = 1024
def read(self, frames):
raise RuntimeError("device disappeared")
assert mic.flush(_Broken()) == 0
-289
View File
@@ -1,289 +0,0 @@
"""End-to-end: microphone in, speech out, over real HTTP.
Every other test in this suite injects a fake at the seam it cares about, and
every one of them passed all week while these got through to production:
- notifications sitting unspoken for minutes (a clock stamped in the wrong
order, two correct units)
- the pet saying the same thing twice (server discarded prose, model repeated
it both sides behaving as written)
- `[laughing]` read out loud (a tag that means something to one model and
nothing to the next one down the pipe)
- a device command written as prose (extractor fine, prompt fine, no marker)
They were all *interaction* bugs. So this one runs the actual pipeline against
a real socket: a threaded HTTP server that speaks the desk protocol, the real
`server_client` doing real requests (including NDJSON streaming), the real
controller loop and state machine. The only fakes are where the hardware is
the mic stream and the speakers because those are the two things a test
genuinely cannot have.
"""
import json
import sys
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
import numpy as np
import pytest
from PySide6.QtWidgets import QApplication
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import controller as controller_mod
from bolt_pet.notifications import Notification
from bolt_pet.state import PetState
_app = QApplication.instance() or QApplication(["test"])
# ── a desk server that actually listens on a port ───────────────────────────
class FakeDesk:
"""Scripted responses, real HTTP. Set `.script` per test."""
def __init__(self):
self.script = {}
self.requests = []
self._server = ThreadingHTTPServer(("127.0.0.1", 0), self._handler())
self._thread = threading.Thread(target=self._server.serve_forever, daemon=True)
self._thread.start()
@property
def url(self) -> str:
host, port = self._server.server_address[:2]
return f"http://{host}:{port}"
def stop(self) -> None:
self._server.shutdown()
self._server.server_close()
def _handler(self):
desk = self
class Handler(BaseHTTPRequestHandler):
def log_message(self, *_args):
pass # the test output is not an access log
def do_GET(self):
path = self.path.split("?")[0]
desk.requests.append(("GET", path))
self._json(desk.script.get(path, {"files": []}))
def do_POST(self):
length = int(self.headers.get("Content-Length") or 0)
body = json.loads(self.rfile.read(length) or b"{}")
path = self.path.split("?")[0]
desk.requests.append(("POST", path, body))
response = desk.script.get(path)
if callable(response):
response = response(body)
if path.endswith("converse_stream"):
self._ndjson(response or [])
else:
self._json(response if response is not None else {"type": "reply", "text": "ok"})
def _json(self, payload):
data = json.dumps(payload).encode()
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(data)))
self.end_headers()
self.wfile.write(data)
def _ndjson(self, events):
self.send_response(200)
self.send_header("Content-Type", "application/x-ndjson")
self.end_headers()
for event in events:
self.wfile.write((json.dumps(event) + "\n").encode())
self.wfile.flush()
return Handler
class FakeMic:
"""Loud frames then quiet ones, so the VAD ends the utterance on its own."""
def __init__(self, loud=8, quiet=60):
self.frames = ([np.full((320, 1), 4000, dtype=np.int16)] * loud
+ [np.zeros((320, 1), dtype=np.int16)] * quiet)
def read(self, _n):
return (self.frames.pop(0) if self.frames
else np.zeros((320, 1), dtype=np.int16)), None
def __enter__(self):
return self
def __exit__(self, *_a):
return False
@pytest.fixture
def pipeline(monkeypatch):
"""A controller wired to a real local server, with fake ears and mouth."""
desk = FakeDesk()
spoken: list[str] = []
levels: list[float] = []
monkeypatch.setattr(controller_mod.config, "SERVER_URL", desk.url)
monkeypatch.setattr(controller_mod.config, "API_KEY", "test-key")
monkeypatch.setattr(controller_mod.config, "SESSION_ID", "pet-smoke")
monkeypatch.setattr(controller_mod.config, "SILENCE_END_SEC", 0.2)
monkeypatch.setattr(controller_mod.config, "MIN_UTTERANCE_S", 0.0)
monkeypatch.setattr(controller_mod.config, "RECEIVE_FILES", False)
# Off by default so each test picks its own path; the streaming tests opt in.
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", False)
monkeypatch.setattr(controller_mod.screen_context, "context_for", lambda text: text)
monkeypatch.setattr(controller_mod.screen_context, "is_fullscreen_active", lambda: False)
# Deepgram and the speakers are the two things a test can't have.
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's on my disk?")
monkeypatch.setattr(controller_mod.stt_stream.StreamingTranscriber, "open",
classmethod(lambda cls, **kw: None))
def fake_speak(text, on_error=None, should_stop=None, voice_id=None, on_level=None):
spoken.append(text)
if on_level is not None:
on_level(0.8) # the mouth opens while a word plays...
levels.append(0.8)
return True
monkeypatch.setattr(controller_mod.tts, "speak", fake_speak)
ctrl = controller_mod.PetController()
ctrl._stream = FakeMic()
try:
yield ctrl, desk, spoken, levels
finally:
desk.stop()
# ── the whole path ──────────────────────────────────────────────────────────
def test_a_plain_turn_goes_mic_to_speaker(pipeline):
ctrl, desk, spoken, _levels = pipeline
desk.script["/desk/converse"] = {"type": "reply", "text": "About sixty percent full."}
ctrl._handle_conversation_turn()
assert spoken == ["About sixty percent full."]
assert ctrl._state.state == PetState.IDLE
posted = [r for r in desk.requests if r[0] == "POST"]
assert posted[0][1] == "/desk/converse"
assert "what's on my disk?" in posted[0][2]["text"]
assert ctrl.history.entries()[-1].text == "About sixty percent full."
def test_a_streamed_turn_speaks_each_sentence_as_it_lands(pipeline, monkeypatch):
"""The NDJSON is parsed by the real client over a real socket — the layer
that a mocked `converse` can never exercise."""
ctrl, desk, spoken, _levels = pipeline
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", True)
desk.script["/desk/converse_stream"] = [
{"type": "say", "text": "The disk is fine."},
{"type": "say", "text": "About sixty percent used."},
{"type": "reply", "text": "The disk is fine. About sixty percent used.",
"already_spoken": True},
]
ctrl._handle_conversation_turn()
assert spoken == ["The disk is fine.", "About sixty percent used."]
# The final reply must not be spoken a third time.
assert len(spoken) == 2
def test_a_tool_turn_says_give_me_a_sec_then_the_answer(pipeline, monkeypatch):
"""Holding line, relayed command, and the real answer — in that order."""
ctrl, desk, spoken, _levels = pipeline
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command",
lambda cmd: ran.append(cmd) or "[exit 0]\n60% used")
desk.script["/desk/converse"] = {
"type": "command", "command": "df -h /", "token": "tok",
"say": "Let me check that for you.",
}
desk.script["/desk/tool_result"] = {"type": "reply", "text": "Sixty percent used."}
ctrl._handle_conversation_turn()
assert spoken == ["Let me check that for you.", "Sixty percent used."]
assert ran == ["df -h /"]
relayed = next(r for r in desk.requests if r[0] == "POST" and r[1] == "/desk/tool_result")
assert relayed[2]["output"].endswith("60% used")
def test_a_device_command_never_reaches_the_shell(pipeline, monkeypatch):
ctrl, desk, spoken, _levels = pipeline
ran = []
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
actions = []
ctrl.action.connect(actions.append)
desk.script["/desk/converse"] = {
"type": "command", "command": "petctl emote wave", "token": "tok"}
desk.script["/desk/tool_result"] = {"type": "reply", "text": "There you go."}
ctrl._handle_conversation_turn()
assert ran == []
assert actions == [{"action": "emote", "emote": "wave"}]
assert spoken == ["There you go."]
def test_the_mouth_moves_with_the_audio(pipeline):
"""Lip-sync is only real if the level actually reaches the window."""
ctrl, desk, _spoken, levels = pipeline
mouth = []
ctrl.mouth.connect(mouth.append)
desk.script["/desk/converse"] = {"type": "reply", "text": "Talking now."}
ctrl._handle_conversation_turn()
assert levels, "TTS was never given a level callback"
assert 0.8 in mouth # opened while speaking...
assert mouth[-1] == 0.0 # ...and closed at the end
def test_a_notification_is_forwarded_and_spoken(pipeline):
"""The path that was silently sitting for five to ten minutes."""
ctrl, desk, spoken, _levels = pipeline
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
desk.script["/desk/converse"] = {"type": "reply", "text": "Harry says he's around."}
ctrl._queue_notification(Notification(app="Signal", summary="Harry", body="you there?"))
ctrl._maybe_heartbeat()
assert spoken == ["Harry says he's around."]
forwarded = next(r for r in desk.requests if r[0] == "POST" and r[1] == "/desk/converse")
assert "Harry" in forwarded[2]["text"]
def test_streaming_falls_back_when_the_server_is_older(pipeline, monkeypatch):
"""A server without /desk/converse_stream (or one that answers with nothing)
must not cost a turn the client drops to the plain endpoint. This is how
the pet keeps working against a container that hasn't been updated yet."""
ctrl, desk, spoken, _levels = pipeline
monkeypatch.setattr(controller_mod.config, "STREAMING_REPLIES", True)
desk.script["/desk/converse_stream"] = [] # nothing streamed back
desk.script["/desk/converse"] = {"type": "reply", "text": "Still here."}
ctrl._handle_conversation_turn()
assert spoken == ["Still here."]
paths = [r[1] for r in desk.requests if r[0] == "POST"]
assert paths == ["/desk/converse_stream", "/desk/converse"]
def test_a_dead_server_leaves_the_pet_usable(pipeline):
"""It should flash an error and go back to listening, not wedge."""
ctrl, desk, spoken, _levels = pipeline
desk.stop() # the server disappears mid-session
states = []
ctrl.state_changed.connect(states.append)
ctrl._handle_conversation_turn()
assert states[-2:] == ["error", "idle"]
assert spoken == []
+51
View File
@@ -1,4 +1,6 @@
import os
import sys
import time
from pathlib import Path
from unittest.mock import MagicMock, patch
@@ -142,3 +144,52 @@ def test_download_outbox_file_raises_server_error_on_http_failure():
get.return_value = _mock_response({}, ok=False)
with pytest.raises(server_client.ServerError, match="abc"):
server_client.download_outbox_file("abc")
def test_converse_reports_a_relay_that_never_produced_a_reply():
"""Hitting the hop cap used to surface as "unknown server response", which
sent everyone looking at the payload shape instead of at a model that kept
calling tools and never answered."""
command = {"type": "command", "command": "echo hi", "token": "t"}
with patch.object(server_client.requests, "post") as post:
post.return_value = _mock_response(command)
with pytest.raises(server_client.ServerError, match="hop cap"):
server_client.converse("hi", on_command=lambda cmd: "ok")
# ── relayed shell commands ──────────────────────────────────────────────────
@pytest.fixture(autouse=True)
def _no_sudo_prompt(monkeypatch):
monkeypatch.setattr(server_client.config, "SUDO_ASKPASS_PROMPT", False)
@pytest.mark.skipif(os.name != "posix", reason="POSIX shell and process groups")
def test_a_successful_command_returns_its_output_and_exit_code():
output = server_client.run_local_command("echo hello; exit 3")
assert output.startswith("[exit 3]")
assert "hello" in output
@pytest.mark.skipif(os.name != "posix", reason="POSIX shell and process groups")
def test_a_timed_out_command_still_reports_what_it_printed():
"""A bare "timed out" tells the model nothing; the last line of output
usually says exactly what it was stuck waiting for."""
output = server_client.run_local_command("echo working on it; sleep 30", timeout=1)
assert "timed out after 1s" in output
assert "working on it" in output
@pytest.mark.skipif(os.name != "posix", reason="POSIX shell and process groups")
def test_a_timed_out_command_takes_its_children_with_it(tmp_path):
"""subprocess.run() would only kill the `sh`, leaving whatever it spawned
running for the rest of the session with no parent watching."""
marker = tmp_path / "ticks"
server_client.run_local_command(
f"(while true; do echo tick >> {marker}; sleep 0.05; done) & sleep 30",
timeout=1,
)
settled = marker.stat().st_size if marker.exists() else 0
time.sleep(0.4)
grew = (marker.stat().st_size if marker.exists() else 0) - settled
assert grew == 0, "a grandchild survived the timeout and is still writing"
-260
View File
@@ -1,260 +0,0 @@
"""The rules about talking — the four kinds of utterance and the mic policy.
These were four near-copies in the controller before, and the copies had
drifted: one didn't arm barge-in, one skipped the follow-up rule. The value of
having one `Speaker` is only real if the differences between the kinds stay
*visible*, so this asserts on the differences rather than on the machinery.
"""
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import config, speech
from bolt_pet.speech import Speaker, Utterance
from bolt_pet.state import PetState, PetStateMachine
class FakeTts:
def __init__(self, completed=True):
self.completed = completed
self.calls = []
def speak(self, text, on_error=None, should_stop=None, voice_id=None, on_level=None):
self.calls.append({"text": text, "voice_id": voice_id,
"interruptible": should_stop is not None})
if on_level is not None:
on_level(0.7)
return self.completed
def play_pcm(self, pcm, sample_rate, should_stop=None, on_level=None):
self.calls.append({"pcm": pcm, "rate": sample_rate,
"interruptible": should_stop is not None})
return self.completed
class FakeBargeIn:
def __init__(self):
self.resets = 0
def reset(self):
self.resets += 1
def check(self, _frame=None):
return False
def build(**kwargs):
"""A speaker plus the things worth asserting on."""
state = PetStateMachine()
tts = kwargs.pop("tts", None) or FakeTts()
said, logged, recorded, levels = [], [], [], []
speaker = Speaker(
state=state, tts=tts,
history=recorded.append,
on_said=said.append,
on_log=logged.append,
on_level=levels.append,
**kwargs,
)
return speaker, state, tts, said, logged, recorded, levels
# ── what distinguishes the four kinds ───────────────────────────────────────
def test_a_reply_is_recorded_but_a_holding_line_is_not():
"""Filler must not push the actual answer out of the transcript."""
speaker, state, _tts, _said, _logged, recorded, _levels = build()
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.holding("Give me a sec."))
speaker.say(Utterance.reply("Sixty percent used."))
assert recorded == ["Sixty percent used."]
def test_a_holding_line_resumes_the_turn_it_interrupted():
"""The turn isn't over — a tool is still running — so it must go back to
THINKING, not drop to IDLE and end the turn."""
speaker, state, _tts, *_ = build()
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.holding("Let me check."))
assert state.state == PetState.THINKING
def test_a_holding_line_cannot_be_talked_over():
"""Cutting off "give me a sec" strands the tool that's already running."""
detector = FakeBargeIn()
speaker, state, tts, *_ = build(barge_in=lambda: detector)
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.holding("One sec."))
assert tts.calls[-1]["interruptible"] is False
speaker.say(Utterance.reply("Done."))
assert tts.calls[-1]["interruptible"] is True
def test_a_streamed_sentence_stays_talking_between_sentences():
"""Otherwise the sprite flickers idle-talking-idle down a long answer."""
speaker, state, *_ = build()
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.stream_chunk("The disk is fine."))
assert state.state == PetState.TALKING
def test_a_scene_is_recorded_because_the_user_heard_it():
speaker, state, _tts, _said, _logged, recorded, _levels = build()
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say_pcm(Utterance.scene("Once upon a time."), pcm=b"\x00\x00", sample_rate=44100)
assert recorded == ["Once upon a time."]
assert state.state == PetState.THINKING
# ── barge-in ordering, which is what the copies got wrong ───────────────────
def test_the_detector_is_reset_after_playback_not_only_before():
"""Playback fed the pet's own voice into the wake model's window. If it
isn't cleared afterwards, the idle listener re-hears the last sentence and
the pet answers itself."""
detector = FakeBargeIn()
speaker, state, *_ = build(barge_in=lambda: detector)
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.reply("Hello there."))
assert detector.resets == 2 # armed before, cleared after
def test_the_barge_in_detail_is_captured_before_the_reset():
"""Read it after the reset and every interruption reports zeroed counters —
which reads like hard evidence and is nothing of the sort."""
detector = FakeBargeIn()
details = ["score 0.81 at frame 12", "score 0.000 at frame 0"]
speaker, state, *_ = build(barge_in=lambda: detector,
detail_of=lambda: details.pop(0))
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.reply("Hello there."))
assert speaker.last_detail == "score 0.81 at frame 12"
def test_a_detector_that_appears_late_is_still_used():
"""The detector is built after the speaker — it needs the mic stream — so
it's read through a callable. Holding a copy is how the two drift apart."""
detector = None
speaker, state, tts, *_ = build(barge_in=lambda: detector)
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.reply("Before."))
assert tts.calls[-1]["interruptible"] is False
detector = FakeBargeIn()
speaker.say(Utterance.reply("After."))
assert tts.calls[-1]["interruptible"] is True
# ── the mouth ───────────────────────────────────────────────────────────────
def test_the_mouth_is_closed_when_the_line_ends():
"""A pet left mid-vowel after the audio stops looks broken."""
speaker, state, _tts, _said, _logged, _recorded, levels = build()
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.reply("Talking."))
assert levels[0] > 0
assert levels[-1] == 0.0
# ── text handling ───────────────────────────────────────────────────────────
def test_an_empty_line_is_not_spoken_at_all():
"""A reply that is nothing but an audio tag reduces to '' — and a silent
bubble with no audio is better than the pet announcing "laughing"."""
speaker, state, tts, said, *_ = build()
assert speaker.say(Utterance.reply("[laughing]")) is True
assert tts.calls == []
assert said == []
assert state.state == PetState.IDLE
def test_the_bubble_gets_display_text_and_tts_gets_the_original():
"""The bubble keeps emoji and drops markdown; TTS does its own stripping
(inside speak(), so every path is covered) and needs the real text."""
speaker, state, tts, said, *_ = build()
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.reply("**Sixty** percent 🎉"))
assert said == ["Sixty percent 🎉"]
assert tts.calls[-1]["text"] == "**Sixty** percent 🎉"
def test_a_voice_override_reaches_tts():
speaker, state, tts, *_ = build(voice_id=lambda: "voice-abc")
state.transition(PetState.LISTENING)
state.transition(PetState.THINKING)
speaker.say(Utterance.reply("In character."))
assert tts.calls[-1]["voice_id"] == "voice-abc"
# ── the follow-up rule ──────────────────────────────────────────────────────
@pytest.fixture
def follow_ups_on(monkeypatch):
monkeypatch.setattr(config, "FOLLOW_UP_LISTEN", True)
monkeypatch.setattr(config, "FOLLOW_UP_MAX_TURNS", 3)
def test_an_interruption_always_reopens_the_mic(follow_ups_on):
"""You talked over it — you are mid-sentence, so it has to listen."""
keep, why = speech.follow_up_decision("Anything else?", completed=False, follow_ups=99)
assert keep is True
assert why == "interrupted"
def test_a_question_keeps_the_mic_open_without_the_wake_word(follow_ups_on):
keep, why = speech.follow_up_decision("Want me to check?", completed=True, follow_ups=0)
assert (keep, why) == (True, "question")
def test_a_statement_ends_the_turn(follow_ups_on):
keep, _why = speech.follow_up_decision("Sixty percent used.", completed=True, follow_ups=0)
assert keep is False
def test_the_cap_stops_a_server_that_ends_every_reply_with_a_question(follow_ups_on):
"""Otherwise mic noise loops it forever."""
assert speech.follow_up_decision("Ok?", completed=True, follow_ups=2)[0] is True
keep, why = speech.follow_up_decision("Ok?", completed=True, follow_ups=3)
assert keep is False
assert "cap" in why
def test_the_rule_can_be_switched_off(monkeypatch):
monkeypatch.setattr(config, "FOLLOW_UP_LISTEN", False)
assert speech.follow_up_decision("Ok?", completed=True, follow_ups=0)[0] is False
+23 -28
View File
@@ -4,8 +4,6 @@
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.speech_text import for_display, for_speech, is_question
@@ -61,11 +59,12 @@ def test_display_keeps_emoji_but_drops_markdown():
assert for_display("* one\n* two") == "• one • two"
def test_is_question_only_fires_on_a_trailing_question():
def test_is_question_fires_when_a_question_mark_appears_anywhere():
assert is_question("Ready to run a command or start a project?")
assert is_question("It's 7:15 AM. Want me to set a timer?")
assert is_question("What time is it? It's 7:15 AM.")
assert is_question("Can you help me with this? I need a quick answer.")
assert not is_question("It's 7:15 AM on July 23, 2026.")
assert not is_question("What time is it? It's 7:15 AM.") # asked in passing
def test_is_question_ignores_trailing_decoration():
@@ -81,31 +80,27 @@ def test_is_question_ignores_question_marks_that_are_not_spoken():
assert not is_question(None)
# ── ElevenLabs v3 delivery tags (observed live 2026-07-31) ──────────────────
# "[laughing] That one came through clean" was spoken as "laughing That one
# came through clean": the tags mean something to the dialogue model inside a
# dialoguectl scene, and nothing at all to the ordinary reply voice.
def test_delivery_tags_are_never_spoken_aloud():
spoken = for_speech("[laughing] That one came through clean.")
assert "laughing" not in spoken
assert spoken.startswith("That one came through clean")
def test_abbreviations_are_worded_instead_of_spelled_out():
"""The periods make these look like sentence boundaries, so the voice reads
them letter by letter ("eee gee")."""
assert for_speech("Use a flag, e.g. --force") == "Use a flag, for example force"
assert for_speech("i.e. the config file") == "that is the config file"
assert for_speech("logs, configs, etc.") == "logs, configs, and so on"
assert for_speech("docker vs. podman") == "docker versus podman"
assert for_speech("Fixed in PR #42") == "Fixed in PR number 42"
def test_the_bubble_does_not_caption_a_laugh_nobody_heard():
assert "[laughing]" not in for_display("[laughing] All good.")
def test_abbreviation_wording_is_word_bounded():
""""vs" inside a word or filename isn't an abbreviation."""
assert "versus" not in for_speech("the vscode window")
assert "versus" not in for_speech("revs per minute")
# A markdown heading has no digit after the hashes, so it's still a heading.
assert for_speech("## Results") == "Results"
@pytest.mark.parametrize("tag", [
"[whispering]", "[cheerfully]", "[sighs]", "[laughs]", "[clears throat]",
"[nervously]", "[excited]", "[pause]", "[shouting]",
])
def test_the_common_tags_are_all_covered(tag):
assert "" == for_speech(tag).strip()
def test_ordinary_bracketed_text_survives():
"""The tags are matched narrowly on purpose — real bracketed content is
part of what the user asked to hear."""
assert "1" in for_speech("See reference [1] for details.")
assert "docs" in for_speech("It's in [the docs] somewhere.")
def test_long_option_dashes_are_dropped_but_hyphens_survive():
assert for_speech("run it with --force") == "run it with force"
assert for_speech("check bolt-pet is up-to-date") == "check bolt-pet is up-to-date"
# The rule line is gone; the full stops are _bullets_to_sentences giving the
# voice a pause where the eye saw a line break.
assert for_speech("one\n---\ntwo") == "one. two."
-173
View File
@@ -1,173 +0,0 @@
"""Streaming speech-to-text: the protocol, and the fallback that makes it safe
to switch on at all.
A fake websocket throughout no network, no Deepgram account, no audio.
"""
import json
import sys
import time
from pathlib import Path
import numpy as np
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet.audio import mic
from bolt_pet.audio.stt_stream import StreamingTranscriber
class FakeSocket:
"""Records what was sent; replays scripted Deepgram frames."""
def __init__(self, messages=(), fail_on_send=False):
self.sent = []
self.closed = False
self.fail_on_send = fail_on_send
self._messages = list(messages)
def send_binary(self, data):
if self.fail_on_send:
raise ConnectionError("socket died")
self.sent.append(data)
def send(self, text):
self.sent.append(text)
def recv(self):
if self._messages:
return self._messages.pop(0)
time.sleep(0.01)
raise ConnectionError("closed")
def close(self):
self.closed = True
def _results(transcript, is_final=True):
return json.dumps({
"type": "Results", "is_final": is_final,
"channel": {"alternatives": [{"transcript": transcript}]},
})
def _frame(value=1000):
return np.full(320, value, dtype=np.int16)
# ── the protocol ────────────────────────────────────────────────────────────
def test_frames_go_up_as_they_are_captured():
socket = FakeSocket([_results("what's the weather")])
session = StreamingTranscriber(socket)
for _ in range(3):
session.feed(_frame())
text = session.finish()
assert len(socket.sent) == 4 # three frames plus the close message
assert text == "what's the weather"
assert socket.closed
def test_only_final_results_are_kept():
"""Interim hypotheses change under you; concatenating them would produce
"what what's what's the what's the weather"."""
socket = FakeSocket([
_results("what's", is_final=False),
_results("what's the", is_final=False),
_results("what's the weather", is_final=True),
])
session = StreamingTranscriber(socket)
assert session.finish() == "what's the weather"
def test_several_final_segments_are_joined():
socket = FakeSocket([_results("turn the lights on"), _results("in the kitchen")])
session = StreamingTranscriber(socket)
assert session.finish() == "turn the lights on in the kitchen"
def test_junk_frames_are_ignored_rather_than_killing_the_reader():
"""An exception on the reader thread would silently end transcription for
the rest of the utterance."""
socket = FakeSocket(["not json at all", '{"type":"Metadata"}',
_results("still works")])
session = StreamingTranscriber(socket)
assert session.finish() == "still works"
def test_an_empty_frame_means_the_socket_closed():
"""websocket-client returns "" from recv() on a closed connection, so it
ends the read loop rather than being treated as a blank transcript."""
socket = FakeSocket([_results("heard this much"), "", _results("never arrives")])
session = StreamingTranscriber(socket)
assert session.finish() == "heard this much"
def test_a_socket_that_dies_mid_utterance_gives_up_quietly():
socket = FakeSocket([], fail_on_send=True)
session = StreamingTranscriber(socket)
session.feed(_frame()) # must not raise — recording carries on
assert session.finish() == ""
# ── opening: failure is an ordinary outcome ────────────────────────────────
def test_open_returns_none_when_it_cannot_connect():
"""None means "the one-shot path will do it", not an error."""
def refuse():
raise OSError("no network")
assert StreamingTranscriber.open(connect=refuse) is None
def test_open_returns_a_session_when_it_can():
session = StreamingTranscriber.open(connect=lambda: FakeSocket([_results("hi")]))
assert session is not None
assert session.finish() == "hi"
def test_streaming_is_off_without_the_switch_or_a_key(monkeypatch):
from bolt_pet.audio import stt_stream
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", False)
assert stt_stream.available() is False
monkeypatch.setattr(stt_stream.config, "STT_STREAMING", True)
monkeypatch.setattr(stt_stream.config, "DEEPGRAM_API_KEY", "")
assert stt_stream.available() is False
# ── the capture hook ────────────────────────────────────────────────────────
class _Stream:
"""Loud frames, then quiet ones, so the VAD ends the utterance."""
def __init__(self, loud=6, quiet=40):
self.frames = ([np.full((320, 1), 3000, dtype=np.int16)] * loud
+ [np.zeros((320, 1), dtype=np.int16)] * quiet)
def read(self, n):
return (self.frames.pop(0) if self.frames
else np.zeros((320, 1), dtype=np.int16)), None
def test_recording_hands_every_speech_frame_to_the_listener():
seen = []
pcm = mic.record_utterance(_Stream(), on_frame=seen.append,
silence_end_sec=0.2, min_utterance_s=0.0)
assert pcm is not None
assert len(seen) >= 6 # every frame of speech was streamed
def test_a_listener_that_throws_cannot_break_the_recording():
"""The fallback is about to need this audio — a dead stream must not cost
the recording too."""
def explode(frame):
raise RuntimeError("stream died")
pcm = mic.record_utterance(_Stream(), on_frame=explode,
silence_end_sec=0.2, min_utterance_s=0.0)
assert pcm is not None and len(pcm) > 0
-84
View File
@@ -1,84 +0,0 @@
"""Remembering where the pet was left — and refusing to when that's a trap."""
import json
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from bolt_pet import window_state
def test_a_saved_position_comes_back(tmp_path):
path = tmp_path / "window.json"
window_state.save(1200, 640, path)
assert window_state.load(path) == (1200, 640)
def test_no_file_yet_is_not_an_error(tmp_path):
assert window_state.load(tmp_path / "nope.json") is None
def test_a_corrupt_file_falls_back_to_the_default_corner(tmp_path):
"""Whatever is in there, the pet still has to start."""
path = tmp_path / "window.json"
for junk in ("", "{", "null", "[]", '{"x": "left"}', '{"y": 3}'):
path.write_text(junk)
assert window_state.load(path) is None
def test_saving_creates_the_cache_directory(tmp_path):
path = tmp_path / "deep" / "nested" / "window.json"
window_state.save(10, 20, path)
assert window_state.load(path) == (10, 20)
def test_an_unwritable_location_is_swallowed(tmp_path):
"""A read-only cache dir is a reason to forget the position, not to crash
on every drag."""
window_state.save(1, 2, tmp_path / "no" / "\0bad" / "window.json")
def test_the_write_is_atomic_and_leaves_no_litter(tmp_path):
path = tmp_path / "window.json"
window_state.save(5, 5, path)
window_state.save(7, 7, path)
assert json.loads(path.read_text()) == {"x": 7, "y": 7}
assert [p.name for p in tmp_path.iterdir()] == ["window.json"]
# ── validating against the screens that exist *now* ─────────────────────────
LAPTOP = (0, 0, 1920, 1080)
EXTERNAL = (1920, 0, 4480, 1440)
def test_a_position_on_a_connected_screen_is_kept():
assert window_state.is_visible_on(1700, 900, 128, [LAPTOP])
def test_a_position_on_an_unplugged_monitor_is_refused():
"""The dangerous case: restoring it faithfully puts the pet somewhere you
can't see or reach."""
assert not window_state.is_visible_on(3000, 700, 128, [LAPTOP])
assert window_state.is_visible_on(3000, 700, 128, [LAPTOP, EXTERNAL])
def test_mostly_off_screen_counts_as_gone():
"""A few pixels of ear poking onto the desktop is not "reachable"."""
assert not window_state.is_visible_on(1910, 500, 128, [LAPTOP])
assert window_state.is_visible_on(1830, 500, 128, [LAPTOP])
def test_touching_an_edge_is_not_overlapping():
assert not window_state.is_visible_on(1920, 0, 128, [LAPTOP])
def test_negative_coordinates_are_fine_when_a_screen_is_there():
"""Monitors left of or above the primary have negative origins."""
left_of_primary = (-1920, 0, 0, 1080)
assert window_state.is_visible_on(-900, 400, 128, [left_of_primary, LAPTOP])
def test_no_screens_at_all_is_not_visible():
assert not window_state.is_visible_on(100, 100, 128, [])