Upload
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
@@ -0,0 +1,32 @@
|
||||
{
|
||||
"permissions": {
|
||||
"allow": [
|
||||
"Read(//Main/Docker-Compose/TMN-API/tmn-api/**)",
|
||||
"Read(//Main/Docker-Compose/TMN-API/tmn-api/desk_client/**)",
|
||||
"Bash(/Main/Docker-Compose/TMN-API/tmn-api/.venv/bin/python3 *)",
|
||||
"Bash(python3 -)",
|
||||
"Bash(.venv/bin/python -c \"import PySide6\")",
|
||||
"Bash(.venv/bin/pytest tests/test_wake_word.py tests/test_controller.py tests/test_state.py -q)",
|
||||
"Bash(.venv/bin/pytest tests/test_wake_word.py tests/test_state.py -q)",
|
||||
"Bash(timeout 20 .venv/bin/python -c \"from bolt_pet.audio import wake_word; print\\('imported ok'\\)\")",
|
||||
"Bash(timeout 40 env QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/test_controller.py -q)",
|
||||
"Bash(timeout 40 env QT_QPA_PLATFORM=offscreen /home/themajesticmagician/Documents/bolt-pet/.venv/bin/pytest /home/themajesticmagician/Documents/bolt-pet/tests/ -q)",
|
||||
"Bash(.venv/bin/pip show *)",
|
||||
"Bash(timeout 60 .venv/bin/python -c \"\nimport numpy as np\nfrom bolt_pet.audio import wake_word\n\nmodel = wake_word._default_model\nframe = np.zeros\\(1280, dtype=np.int16\\)\nscores = model.predict\\(frame\\)\nprint\\('scores:', scores\\)\nmodel.reset\\(\\)\nprint\\('OK — real thunderbolt.onnx model loads and predicts'\\)\n\")",
|
||||
"Bash(timeout 120 .venv/bin/python -m pip install -q \"openwakeword>=0.6\")",
|
||||
"Bash(.venv/bin/python -c \"import openwakeword; print\\('installed at', openwakeword.__file__\\)\")",
|
||||
"Bash(timeout 90 .venv/bin/python -c ' *)",
|
||||
"Bash(cp '/home/themajesticmagician/Documents/kenney_robot-pack/PNG/Side view/robot_greenBody.png' /home/themajesticmagician/Documents/bolt-pet/bolt_pet/assets/sprites/idle/frame_00.png)",
|
||||
"Bash(cp '/home/themajesticmagician/Documents/kenney_robot-pack/PNG/Side view/robot_greenDrive1.png' /home/themajesticmagician/Documents/bolt-pet/bolt_pet/assets/sprites/listening/frame_00.png)",
|
||||
"Bash(cp '/home/themajesticmagician/Documents/kenney_robot-pack/PNG/Side view/robot_greenDrive2.png' /home/themajesticmagician/Documents/bolt-pet/bolt_pet/assets/sprites/listening/frame_01.png)",
|
||||
"Bash(cp '/home/themajesticmagician/Documents/kenney_robot-pack/PNG/Side view/robot_greenDamage1.png' /home/themajesticmagician/Documents/bolt-pet/bolt_pet/assets/sprites/thinking/frame_00.png)",
|
||||
"Bash(cp '/home/themajesticmagician/Documents/kenney_robot-pack/PNG/Side view/robot_greenDamage2.png' /home/themajesticmagician/Documents/bolt-pet/bolt_pet/assets/sprites/thinking/frame_01.png)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/ -q)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python -m pytest tests/ -q -p no:cacheprovider)",
|
||||
"Bash(.venv/bin/python *)",
|
||||
"Bash(python *)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python -m pytest tests/test_controller_features.py -q -p no:cacheprovider)",
|
||||
"Bash(QT_QPA_PLATFORM=offscreen .venv/bin/python -c ' *)"
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,110 @@
|
||||
# Copy this file to .env and fill in your values.
|
||||
|
||||
# ── Bolt server (required) ──────────────────────────────────────────────────
|
||||
# The same server URL + key desk_client/bolt_desk.py and the Android app use.
|
||||
# DESK_API_KEY is the master key from the server's own .env, or a per-user
|
||||
# key minted via the api_key_generate desk marker (see the main repo's
|
||||
# CLAUDE.md "Per-user API keys" section).
|
||||
BOLT_SERVER_URL=http://your-server:5002
|
||||
DESK_API_KEY=
|
||||
|
||||
# Unique per machine/install so sessions don't collide. Defaults to
|
||||
# "pet-<hostname>" if unset.
|
||||
#DESK_SESSION_ID=pet-my-desktop
|
||||
|
||||
# ── Wake word ────────────────────────────────────────────────────────────────
|
||||
# openWakeWord model trained for "thunderbolt" — thunderbolt.onnx ships in
|
||||
# the project root next to this file. Point WAKE_MODEL_FILE at a different
|
||||
# .onnx model to change the phrase (same convention as desk_client's
|
||||
# WAKE_MODEL_FILE / bolt.onnx in the main repo).
|
||||
#WAKE_MODEL_FILE=thunderbolt.onnx
|
||||
#WAKE_WORD_THRESHOLD=0.5
|
||||
#WAKE_CHECK_INTERVAL_SECONDS=1.2
|
||||
|
||||
# ── STT (Deepgram) ──────────────────────────────────────────────────────────
|
||||
DEEPGRAM_API_KEY=
|
||||
#DEEPGRAM_MODEL=nova-3
|
||||
|
||||
# ── TTS (ElevenLabs) — omit to use offline TTS only ─────────────────────────
|
||||
ELEVENLABS_API_KEY=
|
||||
ELEVENLABS_VOICE_ID=
|
||||
#ELEVENLABS_MODEL_ID=eleven_flash_v2
|
||||
#TTS_SAMPLE_RATE=24000
|
||||
|
||||
# ── Audio devices (optional — leave blank for the system default) ──────────
|
||||
#MIC_DEVICE=
|
||||
#SPEAKER_DEVICE=
|
||||
|
||||
# ── VAD tuning (optional) ───────────────────────────────────────────────────
|
||||
#VAD_RMS_THRESHOLD=300
|
||||
#VAD_SILENCE_END_SEC=1.2
|
||||
#VAD_MAX_UTTERANCE_SECONDS=15
|
||||
#VAD_MIN_UTTERANCE_SECONDS=0.4
|
||||
|
||||
# ── Pet window (optional) ───────────────────────────────────────────────────
|
||||
#PET_SIZE=160
|
||||
#PET_START_X=
|
||||
#PET_START_Y=
|
||||
#PET_ALWAYS_ON_TOP=true
|
||||
#IDLE_ANIMATION_FPS=6
|
||||
|
||||
# ── Wandering (optional) — the pet strolls to a random spot while idle ───────
|
||||
#PET_WANDER=true
|
||||
#PET_WANDER_INTERVAL_SECONDS=45 # average pause between strolls (randomized 0.5x-1.5x)
|
||||
#PET_WANDER_SPEED=90 # pixels per second
|
||||
#PET_WANDER_MAX_DISTANCE=600 # cap on a single stroll's length
|
||||
#PET_WANDER_MARGIN=20 # keep this far off the screen edges
|
||||
|
||||
# ── Mouse behaviour (optional) ──────────────────────────────────────────────
|
||||
# Shaped input = the square window's transparent corners stop eating clicks.
|
||||
# Click-through = the pet ignores the mouse entirely (control it from the tray).
|
||||
#PET_SHAPED_INPUT=true
|
||||
#PET_CLICK_THROUGH=false
|
||||
#PET_EDGE_SNAP=true
|
||||
#PET_SNAP_MARGIN=48
|
||||
|
||||
# ── Barge-in (optional) — talk over the pet to cut it off ───────────────────
|
||||
# Threshold defaults to 4x VAD_RMS_THRESHOLD because the mic also hears the
|
||||
# pet's own voice out of the speakers. Raise it if playback self-interrupts.
|
||||
#BARGE_IN=true
|
||||
#BARGE_IN_RMS_THRESHOLD=1200
|
||||
#BARGE_IN_FRAMES=4
|
||||
|
||||
# ── Streaming TTS (optional) — starts talking on the first chunk ────────────
|
||||
#TTS_STREAMING=true
|
||||
|
||||
# ── Screen context (optional) ───────────────────────────────────────────────
|
||||
# Sends the focused window's title along with what you said, so "what's this
|
||||
# error?" has a referent. Text only — no screenshots leave the machine.
|
||||
#SCREEN_CONTEXT=true
|
||||
|
||||
# ── Quiet hours / do-not-disturb (optional) ────────────────────────────────
|
||||
# Comma-separated HH:MM-HH:MM ranges; wrapping past midnight is fine. While
|
||||
# napping the pet dims, stops wandering and makes no proactive noise — the
|
||||
# wake word and click-to-talk still work.
|
||||
#QUIET_HOURS=23:00-08:00
|
||||
#DND_ON_FULLSCREEN=true
|
||||
|
||||
# ── Desktop notification bridge (optional, Linux/D-Bus) ────────────────────
|
||||
# Forwards matching desktop notifications to the server so Bolt can react to
|
||||
# them. Off by default: each forwarded notification is a converse() round
|
||||
# trip. NOTIFICATION_FILTER is a regex over "<app>: <summary> <body>".
|
||||
#NOTIFICATION_BRIDGE=false
|
||||
#NOTIFICATION_FILTER=build|deploy|calendar
|
||||
#NOTIFICATION_MIN_INTERVAL_SECONDS=60
|
||||
|
||||
# ── Push-to-talk (optional) ─────────────────────────────────────────────────
|
||||
# Global hotkey; needs pynput and a session that allows global key hooks
|
||||
# (most Wayland sessions don't). Leave blank to disable.
|
||||
#PUSH_TO_TALK_HOTKEY=ctrl+alt+space
|
||||
|
||||
# ── Wake-word tuning (optional) ─────────────────────────────────────────────
|
||||
# Scores within this margin below the threshold show up as "near misses" in
|
||||
# the tray's wake-word tuner.
|
||||
#WAKE_NEAR_MISS_MARGIN=0.2
|
||||
#WAKE_NEAR_MISS_LIMIT=40
|
||||
|
||||
# ── Misc (optional) ──────────────────────────────────────────────────────────
|
||||
#COMMAND_TIMEOUT_SECONDS=30
|
||||
#HEARTBEAT_INTERVAL_SECONDS=60
|
||||
#HISTORY_LIMIT=100
|
||||
@@ -0,0 +1,5 @@
|
||||
.venv/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.env
|
||||
.pytest_cache/
|
||||
@@ -0,0 +1 @@
|
||||
3.12
|
||||
@@ -0,0 +1,199 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## What this is
|
||||
|
||||
A desktop pet (PySide6 window) that is a voice/click UI on top of an external
|
||||
Bolt server's desk API — same brain, memory, tools, and persona as that
|
||||
server's Discord bot and Linux desk client. This repo has **no import
|
||||
dependency** on the server repo; it's a standalone HTTP client configured via
|
||||
its own `.env`.
|
||||
|
||||
Pipeline: `mic → openWakeWord ("thunderbolt", on-device) / push-to-talk /
|
||||
click → record utterance → Deepgram STT → + active-window context → POST
|
||||
/desk/converse → [server may relay a shell command to run on this machine, or
|
||||
a `petctl` pseudo-command that moves/emotes the pet instead] → reply →
|
||||
ElevenLabs streaming TTS (or offline pyttsx3 fallback) → speakers`, with the
|
||||
pet sprite/speech bubble reflecting state throughout, and playback
|
||||
interruptible by talking over it (barge-in).
|
||||
|
||||
Side channels that let the pet act between turns: the heartbeat (proactive
|
||||
announcements), the desktop notification bridge, and autonomous wandering —
|
||||
all suppressed while it's napping (quiet hours / fullscreen DND).
|
||||
|
||||
## Commands
|
||||
|
||||
```bash
|
||||
# Setup + run (creates .venv and installs requirements.txt on first run)
|
||||
./run.sh # macOS/Linux
|
||||
run.bat # Windows
|
||||
|
||||
# Run tests (no pytest config file — tests self-insert repo root via sys.path).
|
||||
# QT_QPA_PLATFORM=offscreen avoids a QApplication segfault on headless/no-display hosts.
|
||||
QT_QPA_PLATFORM=offscreen .venv/bin/pytest tests/
|
||||
.venv/bin/pytest tests/test_state.py::test_happy_path_transitions # single test
|
||||
|
||||
# Convert a grid sprite sheet into the per-frame-PNG convention sprite.py expects
|
||||
python scripts/slice_spritesheet.py path/to/sheet.png assets/sprites/idle --cols 6 --rows 1
|
||||
```
|
||||
|
||||
There is no lint/build step configured beyond pytest. `cp .env.example .env`
|
||||
and fill in `BOLT_SERVER_URL` / `DESK_API_KEY` (+ `DEEPGRAM_API_KEY`,
|
||||
`ELEVENLABS_API_KEY`) before running — without server config the controller
|
||||
logs a missing-config message and exits its thread instead of starting.
|
||||
|
||||
## Architecture
|
||||
|
||||
- **`config.py`** — loads `.env` from the project root (not via python-dotenv;
|
||||
a small hand-rolled parser matching the server repo's `desk_client/bolt_desk.py`
|
||||
convention) into module-level constants. Everything else reads config from
|
||||
here, never `os.environ` directly.
|
||||
- **`state.py`** — `PetStateMachine`, pure logic with no Qt/audio imports (kept
|
||||
that way deliberately for cheap unit testing). Enforces a transition table;
|
||||
notably `IDLE -> TALKING` is legal directly (no LISTENING/THINKING leg)
|
||||
because the heartbeat can make the pet speak proactively/unprompted.
|
||||
- **`controller.py`** — `PetController(QObject)`, the pipeline orchestrator.
|
||||
Runs on a background `QThread` (wired in `ui/app.py`) so audio I/O/network
|
||||
never blocks the Qt event loop; communicates with the UI only through Qt
|
||||
signals (`state_changed`, `said`, `log`, `action`, `napping`), never touches
|
||||
a `QWidget` directly. Also drives the periodic heartbeat
|
||||
(`_maybe_heartbeat`, gated by `HEARTBEAT_INTERVAL_SECONDS`) which lets the
|
||||
server push proactive spoken announcements between user turns, and on the
|
||||
same tick re-evaluates nap state and drains queued desktop notifications.
|
||||
It owns the live wake-word threshold (`wake_threshold()` is passed to
|
||||
`listen_for_wake_word` as a *callable* so the tray slider takes effect
|
||||
mid-listen) and the conversation `history`.
|
||||
- **`server_client.py`** — HTTP client for the desk API, dependency-free
|
||||
beyond `requests` so it's easy to mock in tests. `converse()` loops relaying
|
||||
server-issued shell commands (`run_local_command`, executed via
|
||||
`subprocess.run(shell=True)` as the desktop user, 30s default timeout) via
|
||||
`/desk/tool_result` until the server sends a final `reply` (capped at
|
||||
`_MAX_RELAY_HOPS`). This is the same "full desktop control" trust model as
|
||||
the server repo's other desk clients — commands only ever originate from
|
||||
the user's own voice/click requests in their own session.
|
||||
- **`audio/`** — `mic.py` (energy-based VAD utterance capture, ported from the
|
||||
server repo's `bolt_desk.py`), `wake_word.py` (openWakeWord `thunderbolt.onnx`
|
||||
detection + `NearMissLog` for threshold tuning — see below), `stt.py`
|
||||
(Deepgram), `tts.py` (ElevenLabs, streaming by default — `stream_pcm()` +
|
||||
`play_stream()` start playback on the first chunk; `chunks_to_int16()`
|
||||
carries odd bytes across HTTP chunk boundaries, without which everything
|
||||
after the first split sample plays as static — falling back to whole-clip
|
||||
PCM then offline `pyttsx3`), `barge_in.py` (`BargeInDetector`: N consecutive
|
||||
loud mic frames while the pet is talking cuts playback and starts the next
|
||||
turn; threshold is deliberately ~4x the VAD one because the mic hears the
|
||||
pet's own voice). Each accepts an injectable stream/model/protocol so tests
|
||||
don't need real audio hardware or a display.
|
||||
- **`pet_actions.py`** — `petctl` pseudo-commands (`petctl move top-left`,
|
||||
`petctl emote wave`, `say`/`wander`/`nap`). The desk API has no "move the
|
||||
pet" payload type and this repo can't change the server, so these ride the
|
||||
existing shell-command relay: `controller._handle_command` parses them and
|
||||
they never reach `subprocess`; anything else is a real shell command exactly
|
||||
as before. Pure parsing; the UI half is `PetWindow.apply_action`.
|
||||
- **`screen_context.py`** — active-window title (xprop/xdotool, Win32,
|
||||
osascript) appended to each utterance via `context_for()`, plus
|
||||
`is_fullscreen_active()` for do-not-disturb. Text only — the desk API takes
|
||||
no images. Every probe is best-effort and returns None/False rather than
|
||||
raising; the parsing is split into pure functions that are tested without a
|
||||
display server.
|
||||
- **`quiet.py`** — quiet-hours spec parsing (`23:00-08:00`, wraps midnight,
|
||||
comma-separated). Napping suppresses *proactive* noise and wandering only;
|
||||
wake word / click / push-to-talk still work.
|
||||
- **`notifications.py`** — Linux/D-Bus notification bridge: tails
|
||||
`dbus-monitor`, parses Notify calls (pure `iter_notifications()`), filters
|
||||
and rate-limits them (`NotificationGate`), and the controller forwards
|
||||
survivors through `converse()`. Off by default — each one is a round trip.
|
||||
- **`history.py`** — rolling transcript (`HISTORY_LIMIT` turns) behind the
|
||||
tray's History window and click-to-copy on the bubble.
|
||||
- **`hotkey.py`** — global push-to-talk via `pynput`; soft-fails with a logged
|
||||
reason (Wayland, missing package, macOS permissions) since the wake word is
|
||||
the primary trigger.
|
||||
- **`speech_text.py`** — sanitizes server replies before they're heard/shown.
|
||||
`for_speech()` (called inside `tts.speak()`, so every path to the speakers is
|
||||
covered) strips markdown, emoji, URLs and stray symbols the voice would read
|
||||
literally ("asterisk asterisk"), turns bullet lists into full sentences, and
|
||||
words a few symbols (`&` → "and"). `for_display()` is the looser version for
|
||||
the speech bubble — markdown syntax gone, emoji kept. Pure string logic, no
|
||||
Qt/audio imports.
|
||||
- **`ui/`** — `app.py` wires `QApplication` + `PetWindow` + `PetTray` + the
|
||||
history/tuner windows + the push-to-talk hotkey + the controller thread
|
||||
together; `pet_window.py` is the frameless/translucent/always-on-top sprite
|
||||
window + speech bubble (non-square frames are centered in the square
|
||||
`PET_SIZE` window, see `paintEvent`), and also owns:
|
||||
- **wandering** — a ~30fps timer walks the window toward a random on-screen
|
||||
target every `PET_WANDER_INTERVAL_SECONDS` (randomized), suppressed
|
||||
whenever the pet is non-IDLE, napping, dragged, or has a bubble up. A
|
||||
commanded `petctl move` overrides all of that except the drag.
|
||||
- **emotes** — `emote_transform()` is pure maths (dx, dy, rotation, scale
|
||||
from a 0..1 progress) kept out of `paintEvent` so the curves are unit
|
||||
tested; every emote must return to the identity transform at progress 1.0
|
||||
or the pet ends up permanently askew.
|
||||
- **shaped input / click-through** — `PET_SHAPED_INPUT` masks the window to
|
||||
the sprite's opaque pixels so the square window's transparent corners stop
|
||||
eating clicks (mask rebuilt only when the frame changes, and pinned to the
|
||||
resting position so a bob/spin doesn't thrash it); `PET_CLICK_THROUGH`
|
||||
makes the pet ignore the mouse entirely.
|
||||
- **edge snapping** (`PET_EDGE_SNAP`) after a drag or a stroll, and **nap
|
||||
dimming** (`set_napping`).
|
||||
`sprite.py` loads `assets/sprites/<state>/*.png` (filename-sorted, looping —
|
||||
currently Kenney's CC0 robot pack, see `assets/sprites/README.md`) and falls
|
||||
back to a procedurally-drawn placeholder blob per state if a folder has no
|
||||
frames; `tray.py` is the system tray menu (talk now / mute / nap / wander /
|
||||
click-through / history / wake-word tuning / quit) — the pet window has no
|
||||
title bar or taskbar entry; `history_window.py` and `wake_tuner.py` are the
|
||||
two dialogs it opens.
|
||||
|
||||
### Wake-word detection
|
||||
|
||||
`audio/wake_word.py` uses a custom-trained openWakeWord model,
|
||||
`thunderbolt.onnx` (ships in the project root), the same way the server
|
||||
repo's `desk_client/bolt_desk.py` uses `bolt.onnx` for "hey bolt" — same
|
||||
runtime (openWakeWord, ONNX inference framework), same per-frame
|
||||
`predict()`/`reset()` loop. Every mic frame is scored; any class score at or
|
||||
above `WAKE_WORD_THRESHOLD` (default `0.5`, in `.env`) counts as a
|
||||
detection. Swap `WAKE_MODEL_FILE` to point at a differently-trained `.onnx`
|
||||
model to change the wake phrase — everything downstream (STT, server call,
|
||||
TTS) is unaffected.
|
||||
|
||||
The threshold is tunable at runtime: the tray's **Wake word tuning…** window
|
||||
(`ui/wake_tuner.py`) shows the peak score seen and a rolling list of near
|
||||
misses (frames within `WAKE_NEAR_MISS_MARGIN` *below* the threshold — i.e.
|
||||
the times it nearly heard you), and its slider is read per frame because
|
||||
`listen_for_wake_word` accepts a callable threshold. Set the threshold just
|
||||
under the peak you can hit reliably, then persist it in `.env`.
|
||||
|
||||
### Testing conventions
|
||||
|
||||
`tests/` covers pure logic only (state machine, wake-word scoring loop, mic
|
||||
VAD, HTTP client against mocks) — nothing there needs real audio hardware or
|
||||
a display. Modules under test are written to accept fake streams/models/
|
||||
`on_command` callables specifically to keep tests hardware-free; follow that
|
||||
pattern (inject a `Protocol`-typed collaborator) rather than mocking at the
|
||||
`sounddevice`/`openwakeword` import boundary when adding new testable logic.
|
||||
`test_controller.py`, `test_controller_features.py`, `test_wander.py` and
|
||||
`test_pet_window_features.py` need a `QApplication`, which segfaults without
|
||||
a display unless run with `QT_QPA_PLATFORM=offscreen`.
|
||||
|
||||
Newer subsystems follow the same rule — the testable part is separated from
|
||||
the part that needs hardware: dbus-monitor output is parsed by a pure
|
||||
`iter_notifications(lines)`, xprop output by pure `parse_xprop_*` functions,
|
||||
HTTP chunk reassembly by `chunks_to_int16`, emote motion by
|
||||
`emote_transform`. Tests that touch the controller monkeypatch
|
||||
`screen_context.context_for` / `is_fullscreen_active`, otherwise they shell
|
||||
out to xprop on a headless box.
|
||||
|
||||
## Security notes
|
||||
|
||||
The server can relay a shell command back to this machine to execute as the
|
||||
desktop user (see `server_client.run_local_command`). This is intentional
|
||||
("full desktop control" for things like "open firefox" or disk checks) and
|
||||
matches the trust model of the server repo's other desk clients. Keep
|
||||
`DESK_API_KEY` private and don't expose the desk API port to the open
|
||||
internet.
|
||||
|
||||
Two newer features widen what leaves this machine, both switchable in `.env`:
|
||||
`SCREEN_CONTEXT` appends the focused window's *title* to each utterance
|
||||
(titles often contain file paths, document names, or subject lines), and
|
||||
`NOTIFICATION_BRIDGE` (off by default) forwards matching desktop
|
||||
notifications to the server. Neither sends screenshots or notification
|
||||
contents you haven't matched with `NOTIFICATION_FILTER`.
|
||||
@@ -0,0 +1,175 @@
|
||||
# Bolt Desktop Pet
|
||||
|
||||
A little animated pet that lives on your desktop and is just a face on top
|
||||
of your Bolt server — same brain, memory, tools, and persona as Discord
|
||||
chat and the Linux desk client. It talks to `ai/desk_api.py` on the server
|
||||
exactly the way `desk_client/bolt_desk.py` does; this project only adds the
|
||||
on-screen pet and swaps Deepgram/ElevenLabs playback to be cross-platform
|
||||
(no `mpv`/`ffplay`/`espeak-ng` subprocess calls — pure `sounddevice`).
|
||||
|
||||
```
|
||||
mic → wake-phrase spotter ("thunderbolt") / hotkey / click → record utterance
|
||||
→ Deepgram STT (+ the focused window's title, for "what's this error?")
|
||||
→ POST /desk/converse on your Bolt server → [server may relay a shell
|
||||
command back to run on THIS machine, or a `petctl` command that moves
|
||||
or emotes the pet] → reply → ElevenLabs streaming TTS → speakers
|
||||
→ shown in a speech bubble + the pet's sprite state (idle/listening/
|
||||
thinking/talking) updates the whole time
|
||||
```
|
||||
|
||||
Nothing is sent to the server until the wake phrase fires, you press the
|
||||
push-to-talk hotkey, or you click the pet — plus, if you turn them on, the
|
||||
heartbeat and the desktop-notification bridge.
|
||||
|
||||
## Why a separate project instead of living in the tmn-api repo
|
||||
|
||||
This runs on your desktop machine, not the server — same relationship as
|
||||
`desk_client/` (Linux) or the Android app, both of which are just clients of
|
||||
the desk API over HTTP. It has no import dependency on the server repo at
|
||||
all, so it can be copied anywhere and configured with its own `.env`.
|
||||
|
||||
## Setup
|
||||
|
||||
1. Copy this whole `bolt-pet/` folder to the machine you want the pet to run
|
||||
on (if that isn't already this machine).
|
||||
2. `cp .env.example .env` and fill in:
|
||||
- `BOLT_SERVER_URL` + `DESK_API_KEY` — same as `desk_client/.env` on the
|
||||
server side. Use the server's master `DESK_API_KEY`, or mint yourself a
|
||||
personal one via the desk-only `api_key_generate` marker (see the main
|
||||
repo's `CLAUDE.md` → "Per-user API keys").
|
||||
- `DEEPGRAM_API_KEY` for STT.
|
||||
- `ELEVENLABS_API_KEY` + `ELEVENLABS_VOICE_ID` for TTS (optional — falls
|
||||
back to offline TTS via `pyttsx3` if omitted or if a request fails).
|
||||
3. Run it:
|
||||
- macOS/Linux: `./run.sh`
|
||||
- Windows: `run.bat`
|
||||
|
||||
Both scripts create a local `.venv` and install `requirements.txt` on
|
||||
first run. On Linux you'll also need system packages for audio:
|
||||
`sudo apt install libportaudio2 espeak-ng`.
|
||||
|
||||
The pet appears near the bottom-right of your screen. It wanders off on its
|
||||
own now and then; drag it anywhere and it tucks itself flush against a nearby
|
||||
screen edge. That position isn't saved across restarts (see Known
|
||||
limitations).
|
||||
|
||||
## Talking to it
|
||||
|
||||
- Say **"thunderbolt"** — detected fully on-device by a custom-trained
|
||||
openWakeWord model (`thunderbolt.onnx`, ships in the project root), the
|
||||
same way the server repo's `desk_client/bolt_desk.py` detects "hey bolt"
|
||||
with `bolt.onnx`. Matches the `DEFAULT_WAKE_WORD` already used for Bolt's
|
||||
Discord voice channels, so it's the same word everywhere.
|
||||
- Or press **Ctrl+Alt+Space** (`PUSH_TO_TALK_HOTKEY`) from anywhere — useful
|
||||
in a noisy room where the wake word misfires. Needs `pynput` and a session
|
||||
that allows global key hooks; most Wayland sessions don't, in which case it
|
||||
logs why at startup and everything else still works.
|
||||
- Or just **click the pet** once (a drag doesn't count as a click).
|
||||
- **Talk over it** to cut a long answer short — the mic stays live while it
|
||||
speaks, and barging in starts your next turn immediately (`BARGE_IN`).
|
||||
- Right-click the tray icon for **Talk now**, **Mute mic**, **Nap**,
|
||||
**Wander around**, **Click through the pet**, **History…**, **Wake word
|
||||
tuning…** and **Quit** — the pet window itself has no title bar or taskbar
|
||||
entry.
|
||||
- **Click the speech bubble** to copy what it just said; the tray's
|
||||
**History…** window keeps the last `HISTORY_LIMIT` turns.
|
||||
|
||||
## What it does on its own
|
||||
|
||||
- **Wanders** the desktop while idle (`PET_WANDER`), stands still while
|
||||
listening/thinking/talking or while a bubble is up.
|
||||
- **Moves and emotes on command.** Bolt can relay `petctl move top-left`,
|
||||
`petctl emote wave|hop|spin|nod|shake`, `petctl say ...`, `petctl wander
|
||||
on|off`, `petctl nap on|off`. These are intercepted here and never reach a
|
||||
shell.
|
||||
- **Naps** during `QUIET_HOURS` (e.g. `23:00-08:00`) or while a fullscreen
|
||||
app is focused (`DND_ON_FULLSCREEN`) — it dims, stops wandering, and makes
|
||||
no proactive noise. It still answers when you speak to it.
|
||||
- **Reacts to desktop notifications** if you turn on `NOTIFICATION_BRIDGE`
|
||||
(Linux/D-Bus) and set a `NOTIFICATION_FILTER` regex — matching
|
||||
notifications get forwarded to the server, so it can tell you the deploy
|
||||
went green. Off by default: each one costs a round trip.
|
||||
|
||||
## Wake-word detection
|
||||
|
||||
`bolt_pet/audio/wake_word.py` feeds every mic frame into `thunderbolt.onnx`
|
||||
via the openWakeWord runtime (ONNX inference) and treats any class score at
|
||||
or above `WAKE_WORD_THRESHOLD` (default `0.5`) as a detection — the exact
|
||||
same per-frame `predict()`/`reset()` pattern as `desk_client/bolt_desk.py`'s
|
||||
main loop. Point `WAKE_MODEL_FILE` in `.env` at a different `.onnx` model to
|
||||
change the wake phrase later without touching any other code.
|
||||
|
||||
If it keeps ignoring you (or firing at the TV), open **Wake word tuning…**
|
||||
from the tray: it shows the peak score while you talk and a rolling list of
|
||||
near misses — frames that scored just under the threshold — and the slider
|
||||
takes effect immediately, mid-listen. Set the threshold just below the peak
|
||||
you can hit reliably, then write it into `.env` as `WAKE_WORD_THRESHOLD`.
|
||||
|
||||
## Project layout
|
||||
|
||||
```
|
||||
bolt_pet/
|
||||
config.py .env loading (same pattern as desk_client/bolt_desk.py)
|
||||
state.py PetState enum + a small transition-checked state machine
|
||||
server_client.py /desk/converse, /desk/tool_result, /desk/report_status
|
||||
controller.py the pipeline: wake word -> STT -> server -> TTS, on a QThread
|
||||
speech_text.py strips markdown/emoji/URLs so the voice never says "asterisk"
|
||||
pet_actions.py petctl move/emote/say/wander/nap parsing
|
||||
screen_context.py active-window title + fullscreen detection
|
||||
quiet.py quiet-hours schedule
|
||||
notifications.py desktop notification bridge (Linux/D-Bus)
|
||||
history.py rolling conversation transcript
|
||||
hotkey.py global push-to-talk (pynput, optional)
|
||||
audio/
|
||||
mic.py input stream + energy-based VAD utterance capture
|
||||
wake_word.py openWakeWord thunderbolt.onnx detection (see above)
|
||||
stt.py Deepgram
|
||||
tts.py ElevenLabs streaming PCM, offline pyttsx3 fallback
|
||||
barge_in.py "you started talking" detector, to cut playback short
|
||||
ui/
|
||||
app.py wires QApplication + window + tray + controller thread together
|
||||
pet_window.py frameless/translucent/always-on-top sprite window + speech bubble
|
||||
sprite.py frame animation loader (see assets/sprites/README.md)
|
||||
tray.py system tray menu
|
||||
history_window.py conversation scrollback (copyable)
|
||||
wake_tuner.py live wake-word threshold + near-miss log
|
||||
assets/sprites/ Kenney robot-pack art (CC0) — see assets/sprites/README.md
|
||||
scripts/
|
||||
slice_spritesheet.py cuts a grid sprite sheet into the per-frame convention
|
||||
tests/ pure-logic unit tests (state machine, wake-phrase
|
||||
matching, HTTP client against mocks) — nothing here
|
||||
needs real audio hardware or a display
|
||||
```
|
||||
|
||||
## Security notes
|
||||
|
||||
Same as `desk_client/bolt_desk.py`: the server can relay a shell command
|
||||
back to this machine ("full desktop control" — "open firefox", "how full is
|
||||
my disk", etc.), which this client executes as your desktop user with a
|
||||
30-second timeout (`COMMAND_TIMEOUT_SECONDS`). That's the same trust model
|
||||
as the Linux desk client and the Android app — commands only ever originate
|
||||
from your own voice/click requests in your own session. Keep `DESK_API_KEY`
|
||||
private; don't expose the desk API port to the open internet.
|
||||
|
||||
`petctl` commands (move/emote/say/wander/nap) are handled inside the pet and
|
||||
never reach a shell, so that channel can't run anything.
|
||||
|
||||
Two features widen what leaves this machine, both off-switchable in `.env`:
|
||||
`SCREEN_CONTEXT=true` (default) appends the focused window's *title* to what
|
||||
you say — titles often contain file paths, document names or email subjects —
|
||||
and `NOTIFICATION_BRIDGE=false` (default) can forward matching desktop
|
||||
notifications. No screenshots or images are ever sent.
|
||||
|
||||
## Known limitations / not-yet-done
|
||||
|
||||
- Pet screen position isn't persisted across restarts.
|
||||
- Wandering is a straight walk to a random point — no Shimeji-style physics,
|
||||
wall-climbing or falling.
|
||||
- Push-to-talk and the notification bridge are platform-limited: the hotkey
|
||||
needs a session that allows global key hooks (most Wayland setups don't),
|
||||
and the notification bridge is Linux/D-Bus only.
|
||||
- Barge-in listens through the same mic that hears the pet's own voice. It
|
||||
wants headphones or a decent gap between speaker and mic; if playback
|
||||
interrupts itself, raise `BARGE_IN_RMS_THRESHOLD` or set `BARGE_IN=false`.
|
||||
- Screen context is the window *title* only — the desk API takes text, so
|
||||
there's no screenshot understanding.
|
||||
@@ -0,0 +1,8 @@
|
||||
"""Entry point: python -m bolt_pet"""
|
||||
|
||||
import sys
|
||||
|
||||
from .ui.app import run
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(run())
|
||||
@@ -0,0 +1,45 @@
|
||||
# Sprite assets
|
||||
|
||||
Art: [Kenney's Robot Pack](https://kenney.nl/assets/robot-pack) (CC0 — no
|
||||
attribution required, credited here anyway), the green side-view robot.
|
||||
Source pack lives at `~/Documents/kenney_robot-pack`; only the frames listed
|
||||
below were copied in.
|
||||
|
||||
Convention the loader (`bolt_pet/ui/sprite.py`) expects:
|
||||
|
||||
```
|
||||
assets/sprites/
|
||||
idle/ frame_00.png robot_greenBody (standing)
|
||||
listening/ frame_00.png, frame_01.png robot_greenDrive1/2 (tracks rolling — "leaning in")
|
||||
thinking/ frame_00.png, frame_01.png robot_greenDamage1/2 (flicker — "processing")
|
||||
talking/ frame_00.png, frame_01.png robot_greenBody, robot_greenJump (bounce)
|
||||
error/ frame_00.png robot_greenHurt
|
||||
```
|
||||
|
||||
- One subfolder per pet state (matches `bolt_pet.state.PetState`).
|
||||
- Any `*.png` filenames work — they're played back in alphabetical-sort
|
||||
order, looping, at `IDLE_ANIMATION_FPS` (see `.env`).
|
||||
- Frames are scaled to fit within `PET_SIZE` (default 160px), keeping aspect
|
||||
ratio, and centered in the (square) pet window — the source art here isn't
|
||||
square, so don't assume it fills the frame edge-to-edge.
|
||||
- A state directory with no frames in it falls back to a small
|
||||
procedurally-drawn placeholder blob (see `_placeholder_frames` in
|
||||
`sprite.py`).
|
||||
|
||||
## Swapping in different art
|
||||
|
||||
Replace any state's PNGs (same alphabetical-order-loops convention) to
|
||||
change its look — no code changes needed. If your source is a single grid
|
||||
spritesheet (rows/cols of frames in one PNG) rather than one-file-per-frame,
|
||||
use `scripts/slice_spritesheet.py` to cut it into this folder-of-frames
|
||||
convention:
|
||||
|
||||
```bash
|
||||
python scripts/slice_spritesheet.py path/to/idle_sheet.png assets/sprites/idle \
|
||||
--cols 6 --rows 1
|
||||
```
|
||||
|
||||
If your format is something else entirely (a single animated GIF/APNG, a
|
||||
Spine/DragonBones skeletal export, an Aseprite `.json` atlas, etc.) — tell me
|
||||
the format and I'll adapt `sprite.py`'s loader rather than making you convert
|
||||
by hand.
|
||||
|
After Width: | Height: | Size: 5.6 KiB |
|
After Width: | Height: | Size: 1.4 KiB |
|
After Width: | Height: | Size: 2.3 KiB |
|
After Width: | Height: | Size: 2.3 KiB |
|
After Width: | Height: | Size: 1.4 KiB |
|
After Width: | Height: | Size: 4.0 KiB |
|
After Width: | Height: | Size: 1.7 KiB |
|
After Width: | Height: | Size: 2.0 KiB |
@@ -0,0 +1,62 @@
|
||||
"""Barge-in: notice that the user started talking *while the pet is talking*
|
||||
so playback can be cut short mid-sentence.
|
||||
|
||||
Deliberately dumber than the utterance VAD in mic.py. The mic hears the pet's
|
||||
own voice coming back out of the speakers, so a single loud frame proves
|
||||
nothing — this requires several consecutive frames well above the normal
|
||||
speech threshold (BARGE_IN_RMS_THRESHOLD defaults to 4x VAD_RMS_THRESHOLD).
|
||||
Takes the same injectable stream shape as mic.record_utterance, so tests feed
|
||||
it fake frames instead of real audio hardware.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
from .. import config
|
||||
from .mic import AudioStream, rms
|
||||
|
||||
|
||||
class BargeInDetector:
|
||||
"""Poll-driven: call check() repeatedly while audio plays. Each call
|
||||
consumes exactly one mic frame (80ms at the default frame length), which
|
||||
is also what paces the playback loop's polling."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
stream: AudioStream,
|
||||
threshold: int = None,
|
||||
required_frames: int = None,
|
||||
frame_len: int = config.FRAME_LEN,
|
||||
):
|
||||
self._stream = stream
|
||||
self._threshold = config.BARGE_IN_RMS_THRESHOLD if threshold is None else threshold
|
||||
self._required = max(1, config.BARGE_IN_FRAMES if required_frames is None else required_frames)
|
||||
self._frame_len = frame_len
|
||||
self._loud_frames = 0
|
||||
|
||||
@property
|
||||
def loud_frames(self) -> int:
|
||||
return self._loud_frames
|
||||
|
||||
def reset(self) -> None:
|
||||
self._loud_frames = 0
|
||||
|
||||
def check(self) -> bool:
|
||||
"""True once the user has been loud for long enough to count as an
|
||||
interruption. Never raises: a mic hiccup mid-playback should not kill
|
||||
the reply, it should just mean "no barge-in this frame"."""
|
||||
try:
|
||||
chunk, _ = self._stream.read(self._frame_len)
|
||||
except Exception:
|
||||
return False
|
||||
frame = np.asarray(chunk)
|
||||
if frame.ndim > 1:
|
||||
frame = frame[:, 0]
|
||||
if frame.size == 0:
|
||||
return False
|
||||
if rms(frame) >= self._threshold:
|
||||
self._loud_frames += 1
|
||||
else:
|
||||
self._loud_frames = 0 # a single thump/cough shouldn't count
|
||||
return self._loud_frames >= self._required
|
||||
@@ -0,0 +1,111 @@
|
||||
"""Mic capture + simple energy-based VAD utterance recording.
|
||||
|
||||
Ported from desk_client/bolt_desk.py's record_utterance() — same tuning
|
||||
knobs, same behavior. Kept independent of any UI/threading model so it can
|
||||
be unit tested by feeding it a fake "stream" object.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import wave
|
||||
from typing import Optional, Protocol
|
||||
|
||||
import numpy as np
|
||||
|
||||
from .. import config
|
||||
|
||||
|
||||
class AudioStream(Protocol):
|
||||
"""Minimal shape of the object record_utterance() needs — matches
|
||||
sounddevice.InputStream's .read(frames) -> (data, overflowed)."""
|
||||
|
||||
def read(self, frames: int): ...
|
||||
|
||||
|
||||
def pcm_to_wav_bytes(pcm: np.ndarray, sample_rate: int = config.SAMPLE_RATE) -> bytes:
|
||||
buf = io.BytesIO()
|
||||
with wave.open(buf, "wb") as wf:
|
||||
wf.setnchannels(1)
|
||||
wf.setsampwidth(2)
|
||||
wf.setframerate(sample_rate)
|
||||
wf.writeframes(pcm.tobytes())
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def rms(frame: np.ndarray) -> float:
|
||||
return float(np.sqrt(np.mean(frame.astype(np.float64) ** 2)))
|
||||
|
||||
|
||||
def record_utterance(
|
||||
stream: AudioStream,
|
||||
should_continue=lambda: True,
|
||||
rms_threshold: int = None,
|
||||
silence_end_sec: float = None,
|
||||
max_utterance_s: float = None,
|
||||
min_utterance_s: float = None,
|
||||
frame_len: int = config.FRAME_LEN,
|
||||
sample_rate: int = config.SAMPLE_RATE,
|
||||
) -> Optional[np.ndarray]:
|
||||
"""Capture one utterance from *stream*: wait for speech to start, stop
|
||||
after trailing silence. Returns None if nothing usable was heard.
|
||||
|
||||
*should_continue* is polled each frame so a caller can cancel recording
|
||||
(e.g. the pet window was closed) without needing threading primitives
|
||||
baked into this function.
|
||||
"""
|
||||
rms_threshold = config.RMS_THRESHOLD if rms_threshold is None else rms_threshold
|
||||
silence_end_sec = config.SILENCE_END_SEC if silence_end_sec is None else silence_end_sec
|
||||
max_utterance_s = config.MAX_UTTERANCE_S if max_utterance_s is None else max_utterance_s
|
||||
min_utterance_s = config.MIN_UTTERANCE_S if min_utterance_s is None else min_utterance_s
|
||||
|
||||
frames: list[np.ndarray] = []
|
||||
started = False
|
||||
silence_frames = 0
|
||||
silence_limit = int(silence_end_sec * sample_rate / frame_len)
|
||||
max_frames = int(max_utterance_s * sample_rate / frame_len)
|
||||
grace_frames = int(4.0 * sample_rate / frame_len) # wait up to 4s for speech to begin
|
||||
waited = 0
|
||||
|
||||
while should_continue():
|
||||
chunk, _ = stream.read(frame_len)
|
||||
frame = np.asarray(chunk)[:, 0].copy()
|
||||
frame_rms = rms(frame)
|
||||
if not started:
|
||||
waited += 1
|
||||
if frame_rms >= rms_threshold:
|
||||
started = True
|
||||
frames.append(frame)
|
||||
elif waited > grace_frames:
|
||||
return None # woke it up but said nothing
|
||||
continue
|
||||
frames.append(frame)
|
||||
if frame_rms < rms_threshold:
|
||||
silence_frames += 1
|
||||
if silence_frames >= silence_limit:
|
||||
break
|
||||
else:
|
||||
silence_frames = 0
|
||||
if len(frames) >= max_frames:
|
||||
break
|
||||
|
||||
if not frames:
|
||||
return None
|
||||
pcm = np.concatenate(frames)
|
||||
if len(pcm) < min_utterance_s * sample_rate:
|
||||
return None
|
||||
return pcm
|
||||
|
||||
|
||||
def open_input_stream():
|
||||
"""Real sounddevice input stream, imported lazily so pure-logic tests
|
||||
(record_utterance with a fake stream) don't need PortAudio installed."""
|
||||
import sounddevice as sd
|
||||
|
||||
return sd.InputStream(
|
||||
samplerate=config.SAMPLE_RATE,
|
||||
channels=1,
|
||||
dtype="int16",
|
||||
blocksize=config.FRAME_LEN,
|
||||
device=config.MIC_DEVICE,
|
||||
)
|
||||
@@ -0,0 +1,42 @@
|
||||
"""Speech-to-text for the actual query, after the wake word fires.
|
||||
|
||||
Deepgram, same as desk_client/bolt_desk.py.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import requests
|
||||
|
||||
from .. import config
|
||||
from .mic import pcm_to_wav_bytes
|
||||
|
||||
|
||||
class SttError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def transcribe(pcm) -> str:
|
||||
if not config.DEEPGRAM_API_KEY:
|
||||
raise SttError("DEEPGRAM_API_KEY is not set")
|
||||
try:
|
||||
response = requests.post(
|
||||
"https://api.deepgram.com/v1/listen",
|
||||
params={"model": config.DEEPGRAM_MODEL, "language": "en", "smart_format": "true"},
|
||||
headers={
|
||||
"Authorization": f"Token {config.DEEPGRAM_API_KEY}",
|
||||
"Content-Type": "audio/wav",
|
||||
},
|
||||
data=pcm_to_wav_bytes(pcm),
|
||||
timeout=30,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except Exception as exc:
|
||||
raise SttError(f"transcription request failed: {exc}") from exc
|
||||
try:
|
||||
return (
|
||||
response.json()
|
||||
.get("results", {}).get("channels", [{}])[0]
|
||||
.get("alternatives", [{}])[0].get("transcript", "")
|
||||
).strip()
|
||||
except Exception as exc:
|
||||
raise SttError(f"couldn't parse transcription response: {exc}") from exc
|
||||
@@ -0,0 +1,163 @@
|
||||
"""Text-to-speech: ElevenLabs, requested as raw PCM so playback is just
|
||||
sounddevice — no external player binary (mpv/ffplay), unlike
|
||||
desk_client/bolt_desk.py which shells out because it only targets Linux.
|
||||
|
||||
Falls back to pyttsx3 (offline, cross-platform: SAPI5 on Windows, NSSpeech
|
||||
on macOS, espeak on Linux) if ElevenLabs isn't configured or the request
|
||||
fails, so the pet can still talk with zero cloud config.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterable, Iterator
|
||||
|
||||
import numpy as np
|
||||
import requests
|
||||
|
||||
from .. import config, speech_text
|
||||
|
||||
|
||||
class TtsError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def synthesize_pcm(text: str) -> tuple[np.ndarray, int]:
|
||||
"""Returns (pcm_int16_mono, sample_rate). Raises TtsError on failure —
|
||||
callers should fall back to speak_offline() rather than treating this
|
||||
as fatal."""
|
||||
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
|
||||
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
||||
try:
|
||||
response = requests.post(
|
||||
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}",
|
||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
||||
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
|
||||
timeout=60,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except Exception as exc:
|
||||
raise TtsError(f"ElevenLabs request failed: {exc}") from exc
|
||||
pcm = np.frombuffer(response.content, dtype=np.int16)
|
||||
if pcm.size == 0:
|
||||
raise TtsError("ElevenLabs returned no audio")
|
||||
return pcm, config.TTS_SAMPLE_RATE
|
||||
|
||||
|
||||
def stream_pcm(text: str, chunk_bytes: int = 4096) -> Iterator[np.ndarray]:
|
||||
"""Same audio as synthesize_pcm(), but yielded as it arrives from
|
||||
ElevenLabs' /stream endpoint so playback can start on the first chunk
|
||||
(~300ms) instead of after the whole clip is synthesized. Raises TtsError
|
||||
before yielding anything if the request itself fails, so callers can fall
|
||||
back cleanly; a mid-stream failure just ends the generator."""
|
||||
if not (config.ELEVENLABS_API_KEY and config.ELEVENLABS_VOICE_ID):
|
||||
raise TtsError("ELEVENLABS_API_KEY / ELEVENLABS_VOICE_ID not set")
|
||||
try:
|
||||
response = requests.post(
|
||||
f"https://api.elevenlabs.io/v1/text-to-speech/{config.ELEVENLABS_VOICE_ID}/stream",
|
||||
headers={"xi-api-key": config.ELEVENLABS_API_KEY},
|
||||
params={"output_format": f"pcm_{config.TTS_SAMPLE_RATE}"},
|
||||
json={"text": text, "model_id": config.ELEVENLABS_MODEL_ID},
|
||||
timeout=60,
|
||||
stream=True,
|
||||
)
|
||||
response.raise_for_status()
|
||||
except Exception as exc:
|
||||
raise TtsError(f"ElevenLabs stream request failed: {exc}") from exc
|
||||
return chunks_to_int16(response.iter_content(chunk_size=chunk_bytes))
|
||||
|
||||
|
||||
def chunks_to_int16(byte_chunks: Iterable[bytes]) -> Iterator[np.ndarray]:
|
||||
"""Reassemble a byte stream into int16 frames. HTTP chunk boundaries fall
|
||||
wherever they like, including *inside* a 16-bit sample, so a trailing odd
|
||||
byte has to be carried into the next chunk — otherwise every chunk after
|
||||
the first is shifted by one byte and plays as static."""
|
||||
carry = b""
|
||||
for chunk in byte_chunks:
|
||||
if not chunk:
|
||||
continue
|
||||
data = carry + chunk
|
||||
usable = len(data) - (len(data) % 2)
|
||||
carry = data[usable:]
|
||||
if usable:
|
||||
yield np.frombuffer(data[:usable], dtype=np.int16)
|
||||
|
||||
|
||||
def play_pcm(pcm: np.ndarray, sample_rate: int, blocking: bool = True, should_stop=None) -> bool:
|
||||
"""Play a whole clip. Returns True if it finished, False if *should_stop*
|
||||
(barge-in) cut it short. *should_stop* is polled while audio plays — each
|
||||
poll consumes one mic frame, which is what paces this loop."""
|
||||
import sounddevice as sd
|
||||
|
||||
sd.play(pcm, samplerate=sample_rate, device=config.SPEAKER_DEVICE)
|
||||
if not blocking:
|
||||
return True
|
||||
if should_stop is None:
|
||||
sd.wait()
|
||||
return True
|
||||
while True:
|
||||
try:
|
||||
if not sd.get_stream().active:
|
||||
break
|
||||
except Exception:
|
||||
break # stream already torn down — playback is over
|
||||
if should_stop():
|
||||
sd.stop()
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def play_stream(chunks: Iterable[np.ndarray], sample_rate: int, should_stop=None) -> bool:
|
||||
"""Play int16 chunks as they arrive. Returns False if interrupted."""
|
||||
import sounddevice as sd
|
||||
|
||||
with sd.OutputStream(
|
||||
samplerate=sample_rate, channels=1, dtype="int16", device=config.SPEAKER_DEVICE
|
||||
) as out:
|
||||
for chunk in chunks:
|
||||
if should_stop is not None and should_stop():
|
||||
# abort() rather than draining: barge-in should stop the voice
|
||||
# now, not at the end of the buffered chunk.
|
||||
out.abort()
|
||||
return False
|
||||
out.write(chunk)
|
||||
return True
|
||||
|
||||
|
||||
def speak_offline(text: str) -> None:
|
||||
try:
|
||||
import pyttsx3
|
||||
except ImportError:
|
||||
return # no TTS available at all — caller already logs the text
|
||||
engine = pyttsx3.init()
|
||||
engine.say(text)
|
||||
engine.runAndWait()
|
||||
|
||||
|
||||
def speak(text: str, on_error=None, should_stop=None) -> bool:
|
||||
"""Speak *text*, preferring streaming ElevenLabs, then whole-clip
|
||||
ElevenLabs, then offline TTS. *on_error*, if given, is called with the
|
||||
exception when ElevenLabs fails (useful for logging) — a fallback still
|
||||
runs either way. Returns False if barge-in interrupted playback.
|
||||
|
||||
The text is sanitized first (speech_text.for_speech): server replies are
|
||||
written for a chat window, and a voice reads markdown/emoji literally
|
||||
("asterisk asterisk"). Sanitizing here rather than at the call sites means
|
||||
every path to the speakers — reply, heartbeat announcement — is covered."""
|
||||
text = speech_text.for_speech(text)
|
||||
if not text:
|
||||
return True
|
||||
if config.TTS_STREAMING:
|
||||
try:
|
||||
return play_stream(stream_pcm(text), config.TTS_SAMPLE_RATE, should_stop=should_stop)
|
||||
except TtsError as exc:
|
||||
if on_error is not None:
|
||||
on_error(exc)
|
||||
try:
|
||||
pcm, sample_rate = synthesize_pcm(text)
|
||||
return play_pcm(pcm, sample_rate, should_stop=should_stop)
|
||||
except TtsError as exc:
|
||||
if on_error is not None:
|
||||
on_error(exc)
|
||||
speak_offline(text)
|
||||
return True
|
||||
@@ -0,0 +1,185 @@
|
||||
"""Wake-word detection via a custom-trained openWakeWord model.
|
||||
|
||||
Uses `thunderbolt.onnx` — trained specifically for "thunderbolt", the same
|
||||
way the main repo's `desk_client/bolt_desk.py` uses `bolt.onnx` for "hey
|
||||
bolt". Same runtime (openWakeWord, ONNX inference), same per-frame
|
||||
predict()/reset() pattern; the only difference is the model file
|
||||
(WAKE_MODEL_FILE) and threshold (WAKE_WORD_THRESHOLD), both configurable via
|
||||
.env if a differently-trained model is swapped in later.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import deque
|
||||
from typing import Callable, Optional, Protocol, Union
|
||||
|
||||
import numpy as np
|
||||
|
||||
from .. import config
|
||||
|
||||
|
||||
class WakeModel(Protocol):
|
||||
def predict(self, frame: np.ndarray) -> dict: ...
|
||||
def reset(self) -> None: ...
|
||||
|
||||
|
||||
class NearMissLog:
|
||||
"""Rolling record of frames that *almost* fired the wake word.
|
||||
|
||||
WAKE_WORD_THRESHOLD is otherwise tuned by guessing at a number in .env
|
||||
and seeing whether the pet ignores you. Keeping the near misses (scores
|
||||
within WAKE_NEAR_MISS_MARGIN below the threshold) turns that into
|
||||
evidence: the tray's wake-word tuner shows what your actual "thunderbolt"
|
||||
scores, so you can set the threshold just under it.
|
||||
|
||||
Pure bookkeeping — the caller supplies timestamps, so it's testable.
|
||||
"""
|
||||
|
||||
def __init__(self, limit: int = None, margin: float = None):
|
||||
self._entries: deque[tuple[float, float, float]] = deque( # (timestamp, score, threshold)
|
||||
maxlen=max(1, config.WAKE_NEAR_MISS_LIMIT if limit is None else limit)
|
||||
)
|
||||
self._margin = config.WAKE_NEAR_MISS_MARGIN if margin is None else margin
|
||||
self._peak = 0.0
|
||||
|
||||
@property
|
||||
def peak(self) -> float:
|
||||
"""Highest score seen since the last reset — the "how close did I
|
||||
get?" readout while you test the wake phrase."""
|
||||
return self._peak
|
||||
|
||||
def observe(self, score: float, threshold: float, timestamp: float) -> bool:
|
||||
"""Record *score*; returns True if it counted as a near miss."""
|
||||
self._peak = max(self._peak, score)
|
||||
if score >= threshold or score < threshold - self._margin:
|
||||
return False
|
||||
self._entries.append((timestamp, score, threshold))
|
||||
return True
|
||||
|
||||
def entries(self) -> list[tuple[float, float, float]]:
|
||||
return list(self._entries)
|
||||
|
||||
def clear(self) -> None:
|
||||
self._entries.clear()
|
||||
self._peak = 0.0
|
||||
|
||||
|
||||
def _construct_model(model_cls, model_path: str):
|
||||
"""openwakeword's Model() constructor keyword has drifted across
|
||||
releases (wakeword_models -> wakeword_model_paths) and some builds
|
||||
reject inference_framework entirely — the main repo's ai/wake_word.py
|
||||
hit the same drift and works around it the same way: try each known
|
||||
calling convention in turn."""
|
||||
attempts = [
|
||||
lambda: model_cls(wakeword_model_paths=[model_path], inference_framework="onnx"),
|
||||
lambda: model_cls(wakeword_model_paths=[model_path]),
|
||||
lambda: model_cls(wakeword_models=[model_path], inference_framework="onnx"),
|
||||
lambda: model_cls(wakeword_models=[model_path]),
|
||||
lambda: model_cls([model_path]),
|
||||
]
|
||||
last_exc: Optional[TypeError] = None
|
||||
for attempt in attempts:
|
||||
try:
|
||||
return attempt()
|
||||
except TypeError as exc:
|
||||
last_exc = exc
|
||||
raise RuntimeError(
|
||||
f"Could not construct openwakeword.Model with any known calling convention "
|
||||
f"(last error: {last_exc})"
|
||||
)
|
||||
|
||||
|
||||
class _OpenWakeWordModel:
|
||||
"""Lazily loads the ONNX model on first use so importing this module
|
||||
(and unit-testing listen_for_wake_word with a fake model) never requires
|
||||
onnxruntime/openwakeword or the model file to be present."""
|
||||
|
||||
def __init__(self):
|
||||
self._model = None
|
||||
|
||||
def _ensure_model(self):
|
||||
if self._model is None:
|
||||
from openwakeword.model import Model
|
||||
# from openwakeword.utils import download_models
|
||||
|
||||
# # The pip package doesn't bundle its melspectrogram/embedding
|
||||
# # feature-extraction sub-models — fetch them once on first use
|
||||
# # (no-op if already cached in openwakeword's own resources dir).
|
||||
# # A non-empty, non-matching model_names list keeps this from
|
||||
# # also pulling every official pretrained wakeword model.
|
||||
# download_models(model_names=["thunderbolt"])
|
||||
|
||||
self._model = _construct_model(Model, config.WAKE_MODEL_PATH)
|
||||
return self._model
|
||||
|
||||
def predict(self, frame: np.ndarray) -> dict:
|
||||
return self._ensure_model().predict(frame)
|
||||
|
||||
def reset(self) -> None:
|
||||
if self._model is not None:
|
||||
self._model.reset()
|
||||
|
||||
|
||||
_default_model = _OpenWakeWordModel()
|
||||
|
||||
|
||||
def listen_for_wake_word(
|
||||
stream,
|
||||
should_continue=lambda: True,
|
||||
model: Optional[WakeModel] = None,
|
||||
threshold: Union[float, Callable[[], float], None] = None,
|
||||
on_tick=None,
|
||||
on_score: Optional[Callable[[float, float], None]] = None,
|
||||
) -> bool:
|
||||
"""Block until the wake word fires (returns True) or *should_continue*
|
||||
goes false (returns False).
|
||||
|
||||
Feeds every frame to *model* (the thunderbolt openWakeWord model by
|
||||
default) and treats any class score >= *threshold* as a detection,
|
||||
resetting the model's internal state afterward so the next call starts
|
||||
clean — same pattern as desk_client/bolt_desk.py's main loop.
|
||||
|
||||
*threshold* may be a number or a zero-argument callable. The callable
|
||||
form exists because this function blocks for minutes at a time: the
|
||||
tray's wake-word tuner slider has to be able to change sensitivity
|
||||
*during* a listen, not only at the start of the next one.
|
||||
|
||||
*on_tick*, if given, is called once per ``WAKE_CHECK_INTERVAL_SECONDS``
|
||||
(not every frame — prediction is cheap enough to run on every frame, but
|
||||
this is the only point control returns to the caller while otherwise
|
||||
blocked here for a possibly long time, so it's how a caller drives
|
||||
periodic work, e.g. the heartbeat/announcement poll in controller.py,
|
||||
during quiet stretches with no wake word).
|
||||
|
||||
*on_score*, if given, gets ``(best_score, threshold)`` every frame — used
|
||||
to log near misses for threshold tuning.
|
||||
"""
|
||||
model = model or _default_model
|
||||
if threshold is None:
|
||||
threshold = config.WAKE_WORD_THRESHOLD
|
||||
resolve_threshold = threshold if callable(threshold) else (lambda: threshold)
|
||||
|
||||
frame_len = config.FRAME_LEN
|
||||
check_every_frames = max(1, int(config.WAKE_CHECK_INTERVAL_SECONDS * config.SAMPLE_RATE / frame_len))
|
||||
frames_since_tick = 0
|
||||
|
||||
while should_continue():
|
||||
chunk, _ = stream.read(frame_len)
|
||||
frame = np.asarray(chunk)[:, 0]
|
||||
scores = model.predict(frame)
|
||||
current_threshold = resolve_threshold()
|
||||
best = max(scores.values()) if scores else 0.0
|
||||
|
||||
frames_since_tick += 1
|
||||
if frames_since_tick >= check_every_frames:
|
||||
frames_since_tick = 0
|
||||
if on_tick is not None:
|
||||
on_tick()
|
||||
|
||||
if on_score is not None:
|
||||
on_score(best, current_threshold)
|
||||
|
||||
if scores and best >= current_threshold:
|
||||
model.reset()
|
||||
return True
|
||||
return False
|
||||
@@ -0,0 +1,192 @@
|
||||
"""Environment configuration for the Bolt desktop pet.
|
||||
|
||||
Same lightweight ".env next to the script" pattern as desk_client/bolt_desk.py
|
||||
in the main tmn-api repo, so this project can be copied anywhere (it does not
|
||||
import anything from that repo) and configured the same way.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import platform
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent.parent # project root (one above bolt_pet/)
|
||||
|
||||
|
||||
def _load_env() -> None:
|
||||
env_path = HERE / ".env"
|
||||
if not env_path.exists():
|
||||
return
|
||||
for line in env_path.read_text(encoding="utf-8").splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
key, _, value = line.partition("=")
|
||||
os.environ.setdefault(key.strip(), value.strip())
|
||||
|
||||
|
||||
_load_env()
|
||||
|
||||
|
||||
def _node_name() -> str:
|
||||
try:
|
||||
return platform.node() or "desktop"
|
||||
except Exception:
|
||||
return "desktop"
|
||||
|
||||
|
||||
# ── server / identity ───────────────────────────────────────────────────────
|
||||
|
||||
SERVER_URL = os.environ.get("BOLT_SERVER_URL", "").rstrip("/")
|
||||
API_KEY = os.environ.get("DESK_API_KEY", "")
|
||||
SESSION_ID = os.environ.get("DESK_SESSION_ID", "pet-" + _node_name())
|
||||
|
||||
# ── wake word ────────────────────────────────────────────────────────────────
|
||||
# openWakeWord model trained specifically for "thunderbolt" — same pattern as
|
||||
# desk_client/bolt_desk.py's WAKE_MODEL_FILE (bolt.onnx / "hey bolt") in the
|
||||
# main repo, resolved relative to the project root so it sits next to .env.
|
||||
|
||||
WAKE_MODEL_PATH = str(HERE / os.environ.get("WAKE_MODEL_FILE", "thunderbolt.onnx"))
|
||||
WAKE_WORD_THRESHOLD = float(os.environ.get("WAKE_WORD_THRESHOLD", "0.5"))
|
||||
# How often (independent of prediction, which runs every frame) the wake
|
||||
# listener yields control back to its caller via on_tick — e.g. the
|
||||
# heartbeat poll in controller.py during quiet stretches with no wake word.
|
||||
WAKE_CHECK_INTERVAL_SECONDS = float(os.environ.get("WAKE_CHECK_INTERVAL_SECONDS", "1.2"))
|
||||
|
||||
# ── STT (Deepgram, same as bolt_desk.py) ────────────────────────────────────
|
||||
|
||||
DEEPGRAM_API_KEY = os.environ.get("DEEPGRAM_API_KEY", "")
|
||||
DEEPGRAM_MODEL = os.environ.get("DEEPGRAM_MODEL", "nova-3")
|
||||
|
||||
# ── TTS (ElevenLabs, requested as raw PCM so playback needs no external
|
||||
# player binary — cross-platform via sounddevice instead of shelling out to
|
||||
# mpv/ffplay like bolt_desk.py does on Linux) ───────────────────────────────
|
||||
|
||||
ELEVENLABS_API_KEY = os.environ.get("ELEVENLABS_API_KEY", "")
|
||||
ELEVENLABS_VOICE_ID = os.environ.get("ELEVENLABS_VOICE_ID", "")
|
||||
ELEVENLABS_MODEL_ID = os.environ.get("ELEVENLABS_MODEL_ID", "eleven_flash_v2")
|
||||
# ElevenLabs PCM output formats are named pcm_<sample_rate>.
|
||||
TTS_SAMPLE_RATE = int(os.environ.get("TTS_SAMPLE_RATE", "24000"))
|
||||
|
||||
# ── mic / VAD (same tuning knobs as bolt_desk.py) ───────────────────────────
|
||||
|
||||
MIC_DEVICE = os.environ.get("MIC_DEVICE", "") or None # sounddevice name/index
|
||||
SPEAKER_DEVICE = os.environ.get("SPEAKER_DEVICE", "") or None
|
||||
|
||||
RMS_THRESHOLD = int(os.environ.get("VAD_RMS_THRESHOLD", "300"))
|
||||
SILENCE_END_SEC = float(os.environ.get("VAD_SILENCE_END_SEC", "1.2"))
|
||||
MAX_UTTERANCE_S = float(os.environ.get("VAD_MAX_UTTERANCE_SECONDS", "15"))
|
||||
MIN_UTTERANCE_S = float(os.environ.get("VAD_MIN_UTTERANCE_SECONDS", "0.4"))
|
||||
|
||||
COMMAND_TIMEOUT_SECONDS = int(os.environ.get("COMMAND_TIMEOUT_SECONDS", "30"))
|
||||
HEARTBEAT_INTERVAL_SECONDS = float(os.environ.get("HEARTBEAT_INTERVAL_SECONDS", "60"))
|
||||
|
||||
# ── barge-in (interrupt playback by talking over it) ────────────────────────
|
||||
# The mic stays live while the pet talks; sustained loud frames cut playback
|
||||
# short. The threshold is deliberately well above VAD_RMS_THRESHOLD because
|
||||
# the mic also hears the pet's own voice through the speakers — raise it
|
||||
# further (or set BARGE_IN=false) if playback keeps interrupting itself.
|
||||
|
||||
BARGE_IN = os.environ.get("BARGE_IN", "true").lower() in ("1", "true", "yes", "on")
|
||||
BARGE_IN_RMS_THRESHOLD = int(os.environ.get("BARGE_IN_RMS_THRESHOLD", str(RMS_THRESHOLD * 4)))
|
||||
BARGE_IN_FRAMES = int(os.environ.get("BARGE_IN_FRAMES", "4")) # consecutive loud frames (80ms each)
|
||||
|
||||
# ── streaming TTS ───────────────────────────────────────────────────────────
|
||||
# ElevenLabs' /stream endpoint + chunked playback: the pet starts talking
|
||||
# after the first PCM chunk instead of after the whole clip is synthesized.
|
||||
|
||||
TTS_STREAMING = os.environ.get("TTS_STREAMING", "true").lower() in ("1", "true", "yes", "on")
|
||||
|
||||
# ── screen context ──────────────────────────────────────────────────────────
|
||||
# Appends the active window's title to what you say, so "what's this error?"
|
||||
# has a referent. The desk API takes text only, so this is a text annotation
|
||||
# (no screenshot upload).
|
||||
|
||||
SCREEN_CONTEXT = os.environ.get("SCREEN_CONTEXT", "true").lower() in ("1", "true", "yes", "on")
|
||||
|
||||
# ── quiet hours / do-not-disturb ────────────────────────────────────────────
|
||||
# Comma-separated HH:MM-HH:MM ranges (wrapping midnight is fine). While
|
||||
# napping the pet dims, stops wandering, and makes no proactive noise —
|
||||
# wake word and click-to-talk still work.
|
||||
|
||||
QUIET_HOURS = os.environ.get("QUIET_HOURS", "")
|
||||
DND_ON_FULLSCREEN = os.environ.get("DND_ON_FULLSCREEN", "true").lower() in ("1", "true", "yes", "on")
|
||||
|
||||
# ── desktop notification bridge ─────────────────────────────────────────────
|
||||
# Mirrors desktop notifications to the server so Bolt can react to them.
|
||||
# Off by default: every forwarded notification costs a converse() round trip.
|
||||
|
||||
NOTIFICATION_BRIDGE = os.environ.get("NOTIFICATION_BRIDGE", "false").lower() in ("1", "true", "yes", "on")
|
||||
# Regex matched against "<app>: <summary> <body>"; empty means "everything".
|
||||
NOTIFICATION_FILTER = os.environ.get("NOTIFICATION_FILTER", "")
|
||||
NOTIFICATION_MIN_INTERVAL_SECONDS = float(os.environ.get("NOTIFICATION_MIN_INTERVAL_SECONDS", "60"))
|
||||
|
||||
# ── conversation history ────────────────────────────────────────────────────
|
||||
|
||||
HISTORY_LIMIT = int(os.environ.get("HISTORY_LIMIT", "100"))
|
||||
|
||||
# ── push-to-talk ────────────────────────────────────────────────────────────
|
||||
# Global hotkey (needs `pynput`; unavailable on most Wayland sessions, in
|
||||
# which case it logs once and the wake word / tray still work). Empty to
|
||||
# disable.
|
||||
|
||||
PUSH_TO_TALK_HOTKEY = os.environ.get("PUSH_TO_TALK_HOTKEY", "ctrl+alt+space")
|
||||
|
||||
# ── wake-word tuning ────────────────────────────────────────────────────────
|
||||
# Scores this far below the threshold are recorded as "near misses" and shown
|
||||
# in the tray's wake-word tuner, so the threshold can be set from evidence.
|
||||
|
||||
WAKE_NEAR_MISS_MARGIN = float(os.environ.get("WAKE_NEAR_MISS_MARGIN", "0.2"))
|
||||
WAKE_NEAR_MISS_LIMIT = int(os.environ.get("WAKE_NEAR_MISS_LIMIT", "40"))
|
||||
|
||||
SAMPLE_RATE = 16000 # mic capture / STT rate
|
||||
FRAME_LEN = 1280 # 80ms @ 16kHz — matches bolt_desk.py's chunking
|
||||
|
||||
# ── pet window ───────────────────────────────────────────────────────────────
|
||||
|
||||
PET_SIZE = int(os.environ.get("PET_SIZE", "160")) # on-screen pixel size (square)
|
||||
PET_START_X = os.environ.get("PET_START_X", "") # blank = bottom-right of primary screen
|
||||
PET_START_Y = os.environ.get("PET_START_Y", "")
|
||||
PET_ALWAYS_ON_TOP = os.environ.get("PET_ALWAYS_ON_TOP", "true").lower() in ("1", "true", "yes", "on")
|
||||
IDLE_ANIMATION_FPS = float(os.environ.get("IDLE_ANIMATION_FPS", "6"))
|
||||
|
||||
# ── wandering ────────────────────────────────────────────────────────────────
|
||||
# The pet strolls to a random spot on its own while idle. Only ever moves when
|
||||
# it's IDLE and not speaking/being dragged, so it never walks out from under a
|
||||
# speech bubble mid-sentence.
|
||||
|
||||
PET_WANDER = os.environ.get("PET_WANDER", "true").lower() in ("1", "true", "yes", "on")
|
||||
# Average seconds of standing still between strolls (each wait is randomized
|
||||
# to 0.5x–1.5x this, so the pet doesn't move on an obvious metronome).
|
||||
PET_WANDER_INTERVAL_SECONDS = float(os.environ.get("PET_WANDER_INTERVAL_SECONDS", "45"))
|
||||
PET_WANDER_SPEED = float(os.environ.get("PET_WANDER_SPEED", "90")) # pixels/second
|
||||
# Cap on how far one stroll can be, so it doesn't teleport across a 4K screen.
|
||||
PET_WANDER_MAX_DISTANCE = float(os.environ.get("PET_WANDER_MAX_DISTANCE", "600"))
|
||||
PET_WANDER_MARGIN = int(os.environ.get("PET_WANDER_MARGIN", "20")) # keep off screen edges
|
||||
|
||||
# ── mouse behaviour ─────────────────────────────────────────────────────────
|
||||
# The sprite is drawn in a square translucent window, so its transparent
|
||||
# corners would otherwise swallow clicks meant for whatever is underneath.
|
||||
# PET_SHAPED_INPUT masks the window's input region to the sprite's own opaque
|
||||
# pixels; PET_CLICK_THROUGH goes further and makes the whole pet ignore the
|
||||
# mouse (tray-only control until you turn it back off).
|
||||
|
||||
PET_SHAPED_INPUT = os.environ.get("PET_SHAPED_INPUT", "true").lower() in ("1", "true", "yes", "on")
|
||||
PET_CLICK_THROUGH = os.environ.get("PET_CLICK_THROUGH", "false").lower() in ("1", "true", "yes", "on")
|
||||
# Snap flush to a screen edge when dropped/parked within this many pixels of it.
|
||||
PET_EDGE_SNAP = os.environ.get("PET_EDGE_SNAP", "true").lower() in ("1", "true", "yes", "on")
|
||||
PET_SNAP_MARGIN = int(os.environ.get("PET_SNAP_MARGIN", "48"))
|
||||
|
||||
|
||||
def is_configured() -> bool:
|
||||
return bool(SERVER_URL and API_KEY)
|
||||
|
||||
|
||||
def missing_config() -> list[str]:
|
||||
missing = []
|
||||
if not SERVER_URL:
|
||||
missing.append("BOLT_SERVER_URL")
|
||||
if not API_KEY:
|
||||
missing.append("DESK_API_KEY")
|
||||
return missing
|
||||
@@ -0,0 +1,348 @@
|
||||
"""Orchestrates the pet's mic -> wake word -> STT -> server -> TTS pipeline.
|
||||
|
||||
Runs on a background QThread (see ui/app.py) so the Qt event loop / window
|
||||
painting is never blocked by audio I/O or network calls. Talks to the UI
|
||||
only through Qt signals (state_changed / said / log / action / napping),
|
||||
which Qt marshals safely across threads — this class never touches a QWidget
|
||||
directly.
|
||||
|
||||
Beyond the core loop it owns the side channels that let the pet act on its
|
||||
own: the heartbeat (proactive announcements), the desktop notification
|
||||
bridge, quiet hours, barge-in, and the live wake-word threshold.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
from typing import Optional
|
||||
|
||||
from PySide6.QtCore import QObject, Signal
|
||||
|
||||
from . import config, history as history_mod, notifications, pet_actions, quiet, screen_context, server_client, speech_text
|
||||
from .audio import barge_in, mic, stt, tts, wake_word
|
||||
from .state import PetState, PetStateMachine
|
||||
|
||||
# How often to re-check whether the pet should be napping. The fullscreen
|
||||
# probe shells out to xprop, so this deliberately isn't every heartbeat tick.
|
||||
_NAP_CHECK_INTERVAL_SECONDS = 10.0
|
||||
|
||||
|
||||
class PetController(QObject):
|
||||
state_changed = Signal(str) # PetState.value
|
||||
said = Signal(str) # text now showing in the speech bubble
|
||||
log = Signal(str)
|
||||
action = Signal(dict) # parsed petctl action for the UI to perform
|
||||
napping = Signal(bool) # quiet hours / fullscreen do-not-disturb
|
||||
finished = Signal()
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self._running = True
|
||||
self._muted = False
|
||||
self._talk_now = threading.Event()
|
||||
self._stream = None
|
||||
self._last_heartbeat = 0.0
|
||||
self._state = PetStateMachine(on_change=self._handle_state_change)
|
||||
|
||||
# Conversation scrollback, shared read-only with the UI's History
|
||||
# window. Append-only from this thread; the UI only ever snapshots it.
|
||||
self.history = history_mod.ConversationHistory(limit=config.HISTORY_LIMIT)
|
||||
|
||||
# Wake-word sensitivity is live-tunable (tray tuner), so it's read
|
||||
# through a callable on every frame rather than captured per listen.
|
||||
self._wake_threshold = config.WAKE_WORD_THRESHOLD
|
||||
self._near_misses = wake_word.NearMissLog()
|
||||
|
||||
self._barge_in: Optional[barge_in.BargeInDetector] = None
|
||||
self._napping = False
|
||||
self._nap_forced: Optional[bool] = None # petctl nap on/off overrides the schedule
|
||||
self._last_nap_check = 0.0
|
||||
|
||||
self._notification_watcher: Optional[notifications.NotificationWatcher] = None
|
||||
self._notification_gate = notifications.NotificationGate(
|
||||
config.NOTIFICATION_FILTER, config.NOTIFICATION_MIN_INTERVAL_SECONDS
|
||||
)
|
||||
self._pending_notifications: list[notifications.Notification] = []
|
||||
self._notification_lock = threading.Lock()
|
||||
|
||||
# ── external controls (safe to call from the Qt/UI thread) ─────────
|
||||
|
||||
def request_talk_now(self) -> None:
|
||||
self._talk_now.set()
|
||||
|
||||
def toggle_mute(self) -> bool:
|
||||
self._muted = not self._muted
|
||||
self.log.emit("Muted." if self._muted else "Unmuted.")
|
||||
return self._muted
|
||||
|
||||
def set_napping(self, napping: Optional[bool]) -> None:
|
||||
"""Force the nap state on/off, or pass None to hand control back to
|
||||
the quiet-hours schedule."""
|
||||
self._nap_forced = napping
|
||||
if napping is not None:
|
||||
self._apply_nap_state(napping)
|
||||
|
||||
def wake_threshold(self) -> float:
|
||||
return self._wake_threshold
|
||||
|
||||
def set_wake_threshold(self, value: float) -> None:
|
||||
self._wake_threshold = min(max(float(value), 0.01), 0.99)
|
||||
|
||||
def wake_stats(self) -> dict:
|
||||
return {"peak": self._near_misses.peak, "near_misses": self._near_misses.entries()}
|
||||
|
||||
def reset_wake_stats(self) -> None:
|
||||
self._near_misses.clear()
|
||||
|
||||
def stop(self) -> None:
|
||||
self._running = False
|
||||
self._talk_now.set() # wake up anything blocked waiting on it
|
||||
if self._notification_watcher is not None:
|
||||
self._notification_watcher.stop()
|
||||
|
||||
# ── internal ─────────────────────────────────────────────────────────
|
||||
|
||||
def _handle_state_change(self, _old: PetState, new: PetState) -> None:
|
||||
self.state_changed.emit(new.value)
|
||||
|
||||
def _should_continue(self) -> bool:
|
||||
return self._running
|
||||
|
||||
def run(self) -> None:
|
||||
"""Thread entry point (connected to QThread.started)."""
|
||||
missing = config.missing_config()
|
||||
if missing:
|
||||
self.log.emit(f"Missing config: {', '.join(missing)} — set them in .env and restart.")
|
||||
self.finished.emit()
|
||||
return
|
||||
try:
|
||||
self._stream = mic.open_input_stream()
|
||||
except Exception as exc:
|
||||
self.log.emit(f"Could not open microphone: {exc}")
|
||||
self.finished.emit()
|
||||
return
|
||||
|
||||
if config.BARGE_IN:
|
||||
self._barge_in = barge_in.BargeInDetector(self._stream)
|
||||
|
||||
with self._stream:
|
||||
try:
|
||||
health = server_client.check_health()
|
||||
self.log.emit(f"Connected to server: {health}")
|
||||
except Exception as exc:
|
||||
self.log.emit(f"Server not reachable yet ({exc}) — will keep trying per-request.")
|
||||
self._start_notification_bridge()
|
||||
self._loop()
|
||||
if self._notification_watcher is not None:
|
||||
self._notification_watcher.stop()
|
||||
self.finished.emit()
|
||||
|
||||
def _loop(self) -> None:
|
||||
while self._running:
|
||||
if self._muted:
|
||||
triggered = self._talk_now.wait(timeout=0.5)
|
||||
if not self._running:
|
||||
return
|
||||
if not triggered:
|
||||
continue
|
||||
self._talk_now.clear()
|
||||
else:
|
||||
if not self._wait_for_wake_or_click():
|
||||
if not self._running:
|
||||
return
|
||||
continue
|
||||
self._handle_conversation_turn()
|
||||
|
||||
def _wait_for_wake_or_click(self) -> bool:
|
||||
"""True once either the wake phrase was heard or a click-to-talk
|
||||
request came in; False on a spurious wakeup (loop again)."""
|
||||
def should_continue() -> bool:
|
||||
return self._running and not self._talk_now.is_set()
|
||||
|
||||
detected = wake_word.listen_for_wake_word(
|
||||
self._stream,
|
||||
should_continue=should_continue,
|
||||
threshold=self.wake_threshold, # callable: the tuner slider is live
|
||||
on_tick=self._maybe_heartbeat,
|
||||
on_score=self._observe_wake_score,
|
||||
)
|
||||
if not self._running:
|
||||
return False
|
||||
if detected:
|
||||
self._talk_now.clear() # in case both fired around the same time
|
||||
return True
|
||||
if self._talk_now.is_set():
|
||||
self._talk_now.clear()
|
||||
return True
|
||||
return False
|
||||
|
||||
def _observe_wake_score(self, score: float, threshold: float) -> None:
|
||||
self._near_misses.observe(score, threshold, time.time())
|
||||
|
||||
def _handle_conversation_turn(self) -> None:
|
||||
self._state.transition(PetState.LISTENING)
|
||||
pcm = mic.record_utterance(self._stream, should_continue=self._should_continue)
|
||||
if pcm is None:
|
||||
self._state.transition(PetState.IDLE)
|
||||
return
|
||||
|
||||
self._state.transition(PetState.THINKING)
|
||||
try:
|
||||
text = stt.transcribe(pcm)
|
||||
except stt.SttError as exc:
|
||||
self.log.emit(f"STT failed: {exc}")
|
||||
self._state.transition(PetState.ERROR)
|
||||
self._state.transition(PetState.IDLE)
|
||||
return
|
||||
if not text:
|
||||
self._state.transition(PetState.IDLE)
|
||||
return
|
||||
self.log.emit(f"You: {text}")
|
||||
self.history.add(history_mod.USER, text, time.time())
|
||||
|
||||
try:
|
||||
# What's focused right now rides along, so "what's this error?"
|
||||
# has a referent without you having to describe the window.
|
||||
reply = server_client.converse(
|
||||
screen_context.context_for(text), on_command=self._handle_command
|
||||
)
|
||||
except server_client.ServerError as exc:
|
||||
self.log.emit(f"Server error: {exc}")
|
||||
self._state.transition(PetState.ERROR)
|
||||
self._state.transition(PetState.IDLE)
|
||||
return
|
||||
|
||||
self._speak(reply)
|
||||
self._state.transition(PetState.IDLE)
|
||||
|
||||
def _handle_command(self, command: str) -> str:
|
||||
"""Server-relayed command. `petctl ...` drives the pet's body and
|
||||
never reaches a shell; everything else is a real command, exactly as
|
||||
before (see the security notes in the README)."""
|
||||
try:
|
||||
action = pet_actions.parse(command)
|
||||
except pet_actions.ActionError as exc:
|
||||
self.log.emit(f"petctl: {exc}")
|
||||
return f"[pet] {exc}"
|
||||
if action is None:
|
||||
return server_client.run_local_command(command)
|
||||
self.log.emit(f"Pet action: {action}")
|
||||
if action["action"] == "nap":
|
||||
self.set_napping(bool(action["enabled"]))
|
||||
self.action.emit(action)
|
||||
return pet_actions.describe(action)
|
||||
|
||||
def _speak(self, text: str) -> None:
|
||||
self._state.transition(PetState.TALKING)
|
||||
# Bubble gets the markdown stripped but emoji kept (it can't render
|
||||
# **bold** but draws emoji fine); tts.speak() does its own, stricter
|
||||
# sanitizing for the voice.
|
||||
self.said.emit(speech_text.for_display(text))
|
||||
self.log.emit(f"Bolt: {text}")
|
||||
self.history.add(history_mod.PET, text, time.time())
|
||||
|
||||
should_stop = None
|
||||
if self._barge_in is not None:
|
||||
self._barge_in.reset()
|
||||
should_stop = self._barge_in.check
|
||||
completed = tts.speak(
|
||||
text,
|
||||
on_error=lambda exc: self.log.emit(f"TTS failed: {exc}"),
|
||||
should_stop=should_stop,
|
||||
)
|
||||
if not completed:
|
||||
# You talked over it — take that as the start of the next turn
|
||||
# rather than making you say the wake word again.
|
||||
self.log.emit("Interrupted — listening.")
|
||||
self._talk_now.set()
|
||||
|
||||
# ── quiet hours / do-not-disturb ─────────────────────────────────────
|
||||
|
||||
def _apply_nap_state(self, napping: bool) -> None:
|
||||
if napping == self._napping:
|
||||
return
|
||||
self._napping = napping
|
||||
self.log.emit("Napping — no proactive noise." if napping else "Awake.")
|
||||
self.napping.emit(napping)
|
||||
|
||||
def _refresh_nap_state(self) -> None:
|
||||
if self._nap_forced is not None:
|
||||
self._apply_nap_state(self._nap_forced)
|
||||
return
|
||||
now = time.monotonic()
|
||||
if now - self._last_nap_check < _NAP_CHECK_INTERVAL_SECONDS:
|
||||
return
|
||||
self._last_nap_check = now
|
||||
napping = quiet.is_quiet(
|
||||
config.QUIET_HOURS,
|
||||
on_error=lambda exc: self.log.emit(f"QUIET_HOURS is malformed ({exc}) — ignoring it."),
|
||||
)
|
||||
if not napping and config.DND_ON_FULLSCREEN:
|
||||
napping = screen_context.is_fullscreen_active()
|
||||
self._apply_nap_state(napping)
|
||||
|
||||
# ── desktop notification bridge ──────────────────────────────────────
|
||||
|
||||
def _start_notification_bridge(self) -> None:
|
||||
if not config.NOTIFICATION_BRIDGE:
|
||||
return
|
||||
watcher = notifications.NotificationWatcher(self._queue_notification)
|
||||
problem = watcher.start()
|
||||
if problem:
|
||||
self.log.emit(problem)
|
||||
return
|
||||
self._notification_watcher = watcher
|
||||
self.log.emit("Notification bridge on.")
|
||||
|
||||
def _queue_notification(self, notification: notifications.Notification) -> None:
|
||||
"""Called on the watcher thread — just queue it; forwarding happens on
|
||||
the pipeline thread where it can't collide with a live conversation."""
|
||||
if not self._notification_gate.should_forward(notification, time.monotonic()):
|
||||
return
|
||||
with self._notification_lock:
|
||||
self._pending_notifications.append(notification)
|
||||
|
||||
def _drain_notifications(self) -> None:
|
||||
with self._notification_lock:
|
||||
pending, self._pending_notifications = self._pending_notifications, []
|
||||
for notification in pending:
|
||||
if not self._running or self._napping:
|
||||
return
|
||||
self.log.emit(f"Notification: {notification.as_text()}")
|
||||
self.history.add(history_mod.SYSTEM, notification.as_text(), time.time())
|
||||
try:
|
||||
reply = server_client.converse(
|
||||
f"[desktop notification] {notification.as_text()}",
|
||||
on_command=self._handle_command,
|
||||
)
|
||||
except server_client.ServerError as exc:
|
||||
self.log.emit(f"Couldn't forward notification: {exc}")
|
||||
return
|
||||
if reply.strip():
|
||||
self._speak(reply)
|
||||
self._state.transition(PetState.IDLE)
|
||||
|
||||
# ── heartbeat ────────────────────────────────────────────────────────
|
||||
|
||||
def _maybe_heartbeat(self) -> None:
|
||||
self._refresh_nap_state()
|
||||
now = time.monotonic()
|
||||
if now - self._last_heartbeat < config.HEARTBEAT_INTERVAL_SECONDS:
|
||||
return
|
||||
self._last_heartbeat = now
|
||||
if self._state.state != PetState.IDLE:
|
||||
return
|
||||
if self._napping:
|
||||
return # quiet hours: still answers when spoken to, just doesn't start
|
||||
self._drain_notifications()
|
||||
if self._state.state != PetState.IDLE:
|
||||
return
|
||||
try:
|
||||
announcement = server_client.report_status()
|
||||
except server_client.ServerError as exc:
|
||||
self.log.emit(f"Heartbeat failed: {exc}")
|
||||
return
|
||||
if announcement:
|
||||
self._speak(announcement)
|
||||
self._state.transition(PetState.IDLE)
|
||||
@@ -0,0 +1,62 @@
|
||||
"""Rolling transcript of the conversation.
|
||||
|
||||
The speech bubble hides itself after a few seconds, which is fine for chat
|
||||
but bad for anything you actually needed to read (a command's output, a
|
||||
number, a URL). This keeps the last HISTORY_LIMIT turns so the tray's
|
||||
History window — and click-to-copy on the bubble — have something to show.
|
||||
|
||||
Pure logic, no Qt: the UI half is ui/history_window.py.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import deque
|
||||
from dataclasses import dataclass
|
||||
from typing import Iterable, Optional
|
||||
|
||||
USER = "you"
|
||||
PET = "bolt"
|
||||
SYSTEM = "system"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Entry:
|
||||
role: str
|
||||
text: str
|
||||
timestamp: Optional[float] = None # time.time(); None when not recorded
|
||||
|
||||
def formatted(self, clock=None) -> str:
|
||||
label = {USER: "You", PET: "Bolt", SYSTEM: "—"}.get(self.role, self.role)
|
||||
stamp = clock(self.timestamp) if (clock and self.timestamp) else None
|
||||
return f"[{stamp}] {label}: {self.text}" if stamp else f"{label}: {self.text}"
|
||||
|
||||
|
||||
class ConversationHistory:
|
||||
def __init__(self, limit: int = 100):
|
||||
self._entries: deque[Entry] = deque(maxlen=max(1, limit))
|
||||
|
||||
def add(self, role: str, text: str, timestamp: Optional[float] = None) -> Optional[Entry]:
|
||||
text = (text or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
entry = Entry(role=role, text=text, timestamp=timestamp)
|
||||
self._entries.append(entry)
|
||||
return entry
|
||||
|
||||
def entries(self) -> list[Entry]:
|
||||
return list(self._entries)
|
||||
|
||||
def last(self, role: Optional[str] = None) -> Optional[Entry]:
|
||||
for entry in reversed(self._entries):
|
||||
if role is None or entry.role == role:
|
||||
return entry
|
||||
return None
|
||||
|
||||
def clear(self) -> None:
|
||||
self._entries.clear()
|
||||
|
||||
def as_text(self, clock=None, entries: Optional[Iterable[Entry]] = None) -> str:
|
||||
return "\n".join(e.formatted(clock) for e in (entries if entries is not None else self._entries))
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self._entries)
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Global push-to-talk hotkey.
|
||||
|
||||
The wake word is the primary trigger, but it misfires in a noisy room and
|
||||
won't fire at all if you're on a call — so there's a keyboard fallback that
|
||||
works even when the pet has no focus (it's a frameless Qt.Tool window with no
|
||||
taskbar entry, so an ordinary QShortcut would never see the key).
|
||||
|
||||
Needs `pynput`, which needs an X11/Win32/macOS input hook: on most Wayland
|
||||
sessions it can't grab global keys, and on macOS it needs Accessibility
|
||||
permission. All of that is a soft failure — start() reports the reason and
|
||||
the wake word / tray keep working.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable, Optional
|
||||
|
||||
# Aliases for the names people actually type in a .env file.
|
||||
_ALIASES = {
|
||||
"control": "ctrl",
|
||||
"ctl": "ctrl",
|
||||
"option": "alt",
|
||||
"opt": "alt",
|
||||
"win": "cmd",
|
||||
"windows": "cmd",
|
||||
"super": "cmd",
|
||||
"meta": "cmd",
|
||||
"command": "cmd",
|
||||
"return": "enter",
|
||||
"escape": "esc",
|
||||
"del": "delete",
|
||||
"ins": "insert",
|
||||
"pgup": "page_up",
|
||||
"pgdn": "page_down",
|
||||
}
|
||||
|
||||
|
||||
class HotkeyError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def to_pynput_spec(spec: str) -> str:
|
||||
""""ctrl+alt+space" -> "<ctrl>+<alt>+<space>" (pynput's GlobalHotKeys
|
||||
syntax: named keys in angle brackets, literal characters bare)."""
|
||||
tokens = [t.strip().lower() for t in (spec or "").split("+")]
|
||||
tokens = [t for t in tokens if t]
|
||||
if not tokens:
|
||||
raise HotkeyError("empty hotkey")
|
||||
parts = []
|
||||
for token in tokens:
|
||||
token = _ALIASES.get(token, token)
|
||||
parts.append(token if len(token) == 1 else f"<{token}>")
|
||||
return "+".join(parts)
|
||||
|
||||
|
||||
class GlobalHotkey:
|
||||
"""Fires *callback* whenever the hotkey is pressed, anywhere. Safe to
|
||||
construct unconditionally — nothing happens until start(), and start()
|
||||
reports failure instead of raising into the UI thread."""
|
||||
|
||||
def __init__(self, spec: str, callback: Callable[[], None]):
|
||||
self.spec = spec
|
||||
self._callback = callback
|
||||
self._listener = None
|
||||
|
||||
@property
|
||||
def running(self) -> bool:
|
||||
return self._listener is not None
|
||||
|
||||
def start(self) -> Optional[str]:
|
||||
"""None on success, otherwise a human-readable reason it's off."""
|
||||
if not (self.spec or "").strip():
|
||||
return None # explicitly disabled — not an error worth reporting
|
||||
try:
|
||||
pynput_spec = to_pynput_spec(self.spec)
|
||||
except HotkeyError as exc:
|
||||
return f"push-to-talk hotkey {self.spec!r} is invalid: {exc}"
|
||||
try:
|
||||
from pynput import keyboard
|
||||
except Exception as exc: # ImportError, or a backend that won't load
|
||||
return f"push-to-talk needs pynput ({exc}) — wake word still works"
|
||||
try:
|
||||
listener = keyboard.GlobalHotKeys({pynput_spec: self._safe_callback})
|
||||
listener.daemon = True
|
||||
listener.start()
|
||||
except Exception as exc:
|
||||
return f"push-to-talk unavailable on this session ({exc}) — wake word still works"
|
||||
self._listener = listener
|
||||
return None
|
||||
|
||||
def _safe_callback(self) -> None:
|
||||
# This runs on pynput's listener thread; an exception there would
|
||||
# silently kill the listener for the rest of the session.
|
||||
try:
|
||||
self._callback()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def stop(self) -> None:
|
||||
if self._listener is not None:
|
||||
try:
|
||||
self._listener.stop()
|
||||
except Exception:
|
||||
pass
|
||||
self._listener = None
|
||||
@@ -0,0 +1,166 @@
|
||||
"""Desktop notification bridge (Linux/D-Bus).
|
||||
|
||||
Lets Bolt react to things that happen without you: a build finishing, a
|
||||
calendar alert, a message arriving. Notifications are tailed from
|
||||
`dbus-monitor`, filtered, rate-limited, and handed to the controller, which
|
||||
forwards them through the normal converse() path — so the pet can say
|
||||
"your deploy just went green" instead of only ever answering questions.
|
||||
|
||||
Off by default (NOTIFICATION_BRIDGE): every forwarded notification is a
|
||||
round trip to the server, and an unfiltered desktop can be very chatty.
|
||||
NOTIFICATION_FILTER (a regex) is the main knob for keeping it useful.
|
||||
|
||||
The dbus-monitor *parsing* is pure and unit tested; only the subprocess
|
||||
plumbing needs a real session bus.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import platform
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import threading
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Iterable, Iterator, Optional
|
||||
|
||||
_STRING_LINE = re.compile(r'^\s*string\s+"(.*)"\s*$')
|
||||
_BLOCK_START = re.compile(r"^(method call|signal|method return|error)\b")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Notification:
|
||||
app: str
|
||||
summary: str
|
||||
body: str
|
||||
|
||||
def as_text(self) -> str:
|
||||
parts = [p for p in (self.summary, self.body) if p]
|
||||
joined = " — ".join(parts)
|
||||
return f"{self.app}: {joined}" if self.app else joined
|
||||
|
||||
|
||||
def iter_notifications(lines: Iterable[str]) -> Iterator[Notification]:
|
||||
"""Pull Notification records out of a `dbus-monitor` line stream.
|
||||
|
||||
A Notify call prints its arguments one per line after the header; the
|
||||
string arguments arrive in the order app_name, app_icon, summary, body
|
||||
(replaces_id is a uint32, so it isn't in the string list). Anything
|
||||
that doesn't look like that is skipped rather than guessed at.
|
||||
"""
|
||||
collecting = False
|
||||
strings: list[str] = []
|
||||
|
||||
def _emit() -> Optional[Notification]:
|
||||
if len(strings) < 3:
|
||||
return None
|
||||
return Notification(app=strings[0].strip(), summary=strings[2].strip(),
|
||||
body=(strings[3].strip() if len(strings) > 3 else ""))
|
||||
|
||||
for line in lines:
|
||||
if _BLOCK_START.match(line):
|
||||
if collecting:
|
||||
notification = _emit()
|
||||
if notification is not None:
|
||||
yield notification
|
||||
collecting = "member=Notify" in line
|
||||
strings = []
|
||||
continue
|
||||
if not collecting:
|
||||
continue
|
||||
match = _STRING_LINE.match(line)
|
||||
if match:
|
||||
strings.append(match.group(1))
|
||||
if collecting:
|
||||
notification = _emit()
|
||||
if notification is not None:
|
||||
yield notification
|
||||
|
||||
|
||||
class NotificationGate:
|
||||
"""Filter + rate limit. Clock is passed in (monotonic seconds) so the
|
||||
rate limiting is testable without sleeping."""
|
||||
|
||||
def __init__(self, pattern: str = "", min_interval: float = 60.0):
|
||||
self._min_interval = max(0.0, min_interval)
|
||||
self._last_forwarded = None
|
||||
self._pattern = None
|
||||
if (pattern or "").strip():
|
||||
try:
|
||||
self._pattern = re.compile(pattern, re.IGNORECASE)
|
||||
except re.error:
|
||||
self._pattern = None # a broken regex shouldn't mute everything
|
||||
|
||||
def matches(self, notification: Notification) -> bool:
|
||||
if self._pattern is None:
|
||||
return True
|
||||
return bool(self._pattern.search(notification.as_text()))
|
||||
|
||||
def should_forward(self, notification: Notification, now: float) -> bool:
|
||||
if not notification.as_text().strip():
|
||||
return False
|
||||
if not self.matches(notification):
|
||||
return False
|
||||
if self._last_forwarded is not None and now - self._last_forwarded < self._min_interval:
|
||||
return False
|
||||
self._last_forwarded = now
|
||||
return True
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
return platform.system() == "Linux" and shutil.which("dbus-monitor") is not None
|
||||
|
||||
|
||||
class NotificationWatcher:
|
||||
"""Tails dbus-monitor on a daemon thread, calling *callback* per
|
||||
notification. Best-effort: if the session bus isn't reachable it reports
|
||||
why via start() and stays off."""
|
||||
|
||||
_ARGS = [
|
||||
"dbus-monitor", "--session",
|
||||
"interface='org.freedesktop.Notifications',member='Notify'",
|
||||
]
|
||||
|
||||
def __init__(self, callback: Callable[[Notification], None]):
|
||||
self._callback = callback
|
||||
self._process: Optional[subprocess.Popen] = None
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
self._running = False
|
||||
|
||||
def start(self) -> Optional[str]:
|
||||
"""None on success, otherwise the reason the bridge is off."""
|
||||
if not available():
|
||||
return "notification bridge needs Linux + dbus-monitor — skipping"
|
||||
try:
|
||||
self._process = subprocess.Popen(
|
||||
self._ARGS, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
|
||||
text=True, bufsize=1,
|
||||
)
|
||||
except Exception as exc:
|
||||
return f"couldn't start dbus-monitor ({exc}) — notification bridge off"
|
||||
self._running = True
|
||||
self._thread = threading.Thread(target=self._pump, name="notification-bridge", daemon=True)
|
||||
self._thread.start()
|
||||
return None
|
||||
|
||||
def _pump(self) -> None:
|
||||
assert self._process is not None and self._process.stdout is not None
|
||||
try:
|
||||
for notification in iter_notifications(self._process.stdout):
|
||||
if not self._running:
|
||||
return
|
||||
try:
|
||||
self._callback(notification)
|
||||
except Exception:
|
||||
pass # one bad notification shouldn't end the bridge
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def stop(self) -> None:
|
||||
self._running = False
|
||||
if self._process is not None:
|
||||
try:
|
||||
self._process.terminate()
|
||||
except Exception:
|
||||
pass
|
||||
self._process = None
|
||||
@@ -0,0 +1,137 @@
|
||||
"""Commands that drive the pet's *body* instead of the shell.
|
||||
|
||||
The server relays shell commands to this machine (see server_client.
|
||||
run_local_command). Rather than inventing a new payload type the desk API
|
||||
doesn't speak — this client can't change the server — a small `petctl`
|
||||
pseudo-command is intercepted before it ever reaches `subprocess`: if Bolt
|
||||
emits `petctl move top-left` or `petctl emote wave`, the pet does it and
|
||||
returns a normal-looking command output string, so from the server's side
|
||||
it's just another tool call that worked.
|
||||
|
||||
Pure parsing logic — no Qt, no subprocess — so it's cheap to unit test. The
|
||||
UI half lives in ui/pet_window.py (apply_action).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shlex
|
||||
from typing import Optional
|
||||
|
||||
# What Bolt is allowed to type. Anything else falls through to a real shell.
|
||||
_PREFIXES = ("petctl", "bolt-pet", "pet")
|
||||
|
||||
ANCHORS = (
|
||||
"top-left", "top", "top-right",
|
||||
"left", "center", "right",
|
||||
"bottom-left", "bottom", "bottom-right",
|
||||
"cursor", "random",
|
||||
)
|
||||
|
||||
EMOTES = ("wave", "hop", "spin", "nod", "shake", "bounce", "wiggle")
|
||||
|
||||
HELP = (
|
||||
"petctl move <x> <y> | <" + "|".join(ANCHORS) + ">\n"
|
||||
"petctl emote <" + "|".join(EMOTES) + ">\n"
|
||||
"petctl say <text>\n"
|
||||
"petctl wander on|off\n"
|
||||
"petctl nap on|off"
|
||||
)
|
||||
|
||||
|
||||
class ActionError(Exception):
|
||||
"""Bad petctl syntax — reported back to the server as command output."""
|
||||
|
||||
|
||||
def is_pet_command(command: str) -> bool:
|
||||
parts = (command or "").strip().split()
|
||||
return bool(parts) and parts[0].lower() in _PREFIXES
|
||||
|
||||
|
||||
def _bool_arg(value: str) -> bool:
|
||||
value = value.lower()
|
||||
if value in ("on", "true", "yes", "1", "start", "enable"):
|
||||
return True
|
||||
if value in ("off", "false", "no", "0", "stop", "disable"):
|
||||
return False
|
||||
raise ActionError(f"expected on/off, got {value!r}")
|
||||
|
||||
|
||||
def parse(command: str) -> Optional[dict]:
|
||||
"""Parse a `petctl ...` string into an action dict, or None if this isn't
|
||||
a pet command at all (caller should run it as a real shell command).
|
||||
Raises ActionError on a pet command that doesn't make sense."""
|
||||
if not is_pet_command(command):
|
||||
return None
|
||||
try:
|
||||
parts = shlex.split(command.strip())
|
||||
except ValueError as exc: # unbalanced quotes
|
||||
raise ActionError(f"couldn't parse arguments: {exc}") from exc
|
||||
verb = (parts[1].lower() if len(parts) > 1 else "help")
|
||||
args = parts[2:]
|
||||
|
||||
if verb in ("help", "-h", "--help"):
|
||||
return {"action": "help"}
|
||||
|
||||
if verb in ("move", "goto", "walk"):
|
||||
if not args:
|
||||
raise ActionError("move needs a target: " + ", ".join(ANCHORS) + ", or x y")
|
||||
if len(args) >= 2 and _looks_numeric(args[0]) and _looks_numeric(args[1]):
|
||||
return {"action": "move", "x": int(float(args[0])), "y": int(float(args[1]))}
|
||||
anchor = args[0].lower().replace("_", "-")
|
||||
if anchor not in ANCHORS:
|
||||
raise ActionError(f"unknown position {args[0]!r}; try one of: " + ", ".join(ANCHORS))
|
||||
return {"action": "move", "anchor": anchor}
|
||||
|
||||
if verb in ("emote", "do"):
|
||||
if not args:
|
||||
raise ActionError("emote needs a name: " + ", ".join(EMOTES))
|
||||
emote = args[0].lower()
|
||||
if emote not in EMOTES:
|
||||
raise ActionError(f"unknown emote {args[0]!r}; try one of: " + ", ".join(EMOTES))
|
||||
return {"action": "emote", "emote": emote}
|
||||
|
||||
if verb == "say":
|
||||
text = " ".join(args).strip()
|
||||
if not text:
|
||||
raise ActionError("say needs something to say")
|
||||
return {"action": "say", "text": text}
|
||||
|
||||
if verb == "wander":
|
||||
if not args:
|
||||
raise ActionError("wander needs on or off")
|
||||
return {"action": "wander", "enabled": _bool_arg(args[0])}
|
||||
|
||||
if verb in ("nap", "sleep", "dnd"):
|
||||
if not args:
|
||||
raise ActionError("nap needs on or off")
|
||||
return {"action": "nap", "enabled": _bool_arg(args[0])}
|
||||
|
||||
raise ActionError(f"unknown petctl verb {verb!r}\n{HELP}")
|
||||
|
||||
|
||||
def _looks_numeric(value: str) -> bool:
|
||||
try:
|
||||
float(value)
|
||||
return True
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
def describe(action: dict) -> str:
|
||||
"""The text handed back to the server as this "command"'s output. Phrased
|
||||
as a completed fact so the model doesn't narrate the mechanics of it."""
|
||||
kind = action.get("action")
|
||||
if kind == "move":
|
||||
where = action.get("anchor") or f"({action.get('x')}, {action.get('y')})"
|
||||
return f"[pet] walking to {where}"
|
||||
if kind == "emote":
|
||||
return f"[pet] {action['emote']}"
|
||||
if kind == "say":
|
||||
return "[pet] showing that in the speech bubble"
|
||||
if kind == "wander":
|
||||
return "[pet] wandering " + ("enabled" if action["enabled"] else "disabled")
|
||||
if kind == "nap":
|
||||
return "[pet] " + ("napping" if action["enabled"] else "awake")
|
||||
if kind == "help":
|
||||
return HELP
|
||||
return "[pet] ok"
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Quiet hours — when the pet is allowed to make noise on its own.
|
||||
|
||||
Napping only ever suppresses *proactive* noise (heartbeat announcements,
|
||||
forwarded notifications) and wandering. The wake word, click-to-talk and
|
||||
push-to-talk still work: telling it to be quiet shouldn't mean it stops
|
||||
answering when spoken to.
|
||||
|
||||
Pure logic (parsing + a time comparison) so it's testable without waiting
|
||||
for 11pm.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import time as dtime
|
||||
from typing import Iterable, Optional
|
||||
|
||||
|
||||
class QuietHoursError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _parse_clock(value: str) -> dtime:
|
||||
parts = value.strip().split(":")
|
||||
if len(parts) != 2:
|
||||
raise QuietHoursError(f"expected HH:MM, got {value!r}")
|
||||
try:
|
||||
hour, minute = int(parts[0]), int(parts[1])
|
||||
except ValueError as exc:
|
||||
raise QuietHoursError(f"expected HH:MM, got {value!r}") from exc
|
||||
if not (0 <= hour <= 23 and 0 <= minute <= 59):
|
||||
raise QuietHoursError(f"{value!r} is not a real time of day")
|
||||
return dtime(hour, minute)
|
||||
|
||||
|
||||
def parse_ranges(spec: str) -> list[tuple[dtime, dtime]]:
|
||||
""""23:00-08:00, 13:00-14:00" -> [(23:00, 08:00), (13:00, 14:00)].
|
||||
Empty/blank spec means "no quiet hours"."""
|
||||
ranges: list[tuple[dtime, dtime]] = []
|
||||
for chunk in (spec or "").split(","):
|
||||
chunk = chunk.strip()
|
||||
if not chunk:
|
||||
continue
|
||||
start, sep, end = chunk.partition("-")
|
||||
if not sep:
|
||||
raise QuietHoursError(f"expected HH:MM-HH:MM, got {chunk!r}")
|
||||
ranges.append((_parse_clock(start), _parse_clock(end)))
|
||||
return ranges
|
||||
|
||||
|
||||
def in_ranges(now: dtime, ranges: Iterable[tuple[dtime, dtime]]) -> bool:
|
||||
for start, end in ranges:
|
||||
if start == end:
|
||||
continue # a zero-length range is a typo, not "all day"
|
||||
if start < end:
|
||||
if start <= now < end:
|
||||
return True
|
||||
elif now >= start or now < end: # wraps past midnight
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_quiet(spec: str, now: Optional[dtime] = None, on_error=None) -> bool:
|
||||
"""True if *now* (defaults to the local wall clock) falls inside *spec*.
|
||||
A malformed spec is reported via *on_error* and treated as "not quiet" —
|
||||
a config typo shouldn't silently mute the pet forever."""
|
||||
if not (spec or "").strip():
|
||||
return False
|
||||
try:
|
||||
ranges = parse_ranges(spec)
|
||||
except QuietHoursError as exc:
|
||||
if on_error is not None:
|
||||
on_error(exc)
|
||||
return False
|
||||
if now is None:
|
||||
from datetime import datetime
|
||||
|
||||
now = datetime.now().time()
|
||||
return in_ranges(now, ranges)
|
||||
@@ -0,0 +1,184 @@
|
||||
"""What's on screen right now — the active window's title, and whether
|
||||
something is running fullscreen.
|
||||
|
||||
Two consumers:
|
||||
* annotate() tacks the focused window's title onto what you said, so
|
||||
"what's this error?" has a referent without you describing it (the desk
|
||||
API takes text only, so this is a text annotation — no screenshot upload).
|
||||
* is_fullscreen_active() feeds do-not-disturb: the pet shouldn't announce
|
||||
anything over a call or a fullscreen game.
|
||||
|
||||
Everything here is best-effort and must never raise: on a locked-down Wayland
|
||||
session none of it is available, and the correct behaviour is simply "no
|
||||
context", not a crashed pipeline. The subprocess *parsing* is split into pure
|
||||
functions so it can be tested without a display server.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import platform
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
from typing import Optional
|
||||
|
||||
_TIMEOUT = 2.0
|
||||
_MAX_TITLE_CHARS = 160
|
||||
|
||||
# Titles that are just the desktop itself — annotating with these is noise.
|
||||
_BORING_TITLES = {"", "desktop", "@!0,0;bdib", "plasmashell", "gnome-shell", "xfdesktop"}
|
||||
|
||||
|
||||
def _run(args: list[str]) -> Optional[str]:
|
||||
try:
|
||||
completed = subprocess.run(args, capture_output=True, text=True, timeout=_TIMEOUT)
|
||||
except Exception:
|
||||
return None
|
||||
if completed.returncode != 0:
|
||||
return None
|
||||
return completed.stdout
|
||||
|
||||
|
||||
# ── pure parsing helpers (unit tested; no display server needed) ─────────────
|
||||
|
||||
def parse_xprop_window_id(output: str) -> Optional[str]:
|
||||
"""`xprop -root _NET_ACTIVE_WINDOW` -> '_NET_ACTIVE_WINDOW(WINDOW): window id # 0x3c00007'"""
|
||||
match = re.search(r"(0x[0-9a-fA-F]+)", output or "")
|
||||
if not match or int(match.group(1), 16) == 0:
|
||||
return None
|
||||
return match.group(1)
|
||||
|
||||
|
||||
def parse_xprop_window_name(output: str) -> Optional[str]:
|
||||
"""`xprop -id <id> _NET_WM_NAME` -> '_NET_WM_NAME(UTF8_STRING) = "Firefox"'"""
|
||||
match = re.search(r'=\s*"(.*)"\s*$', (output or "").strip(), re.DOTALL)
|
||||
if not match:
|
||||
return None
|
||||
return match.group(1).strip()
|
||||
|
||||
|
||||
def parse_xprop_fullscreen(output: str) -> bool:
|
||||
return "_NET_WM_STATE_FULLSCREEN" in (output or "")
|
||||
|
||||
|
||||
def clean_title(title: Optional[str]) -> Optional[str]:
|
||||
title = (title or "").strip().replace("\n", " ")
|
||||
if title.lower() in _BORING_TITLES:
|
||||
return None
|
||||
if len(title) > _MAX_TITLE_CHARS:
|
||||
title = title[: _MAX_TITLE_CHARS - 1].rstrip() + "…"
|
||||
return title or None
|
||||
|
||||
|
||||
def annotate(text: str, title: Optional[str]) -> str:
|
||||
"""Attach the window title as an explicit aside rather than splicing it
|
||||
into the sentence, so the model can ignore it when it's irrelevant."""
|
||||
text = (text or "").strip()
|
||||
title = clean_title(title)
|
||||
if not text or not title:
|
||||
return text
|
||||
return f"{text}\n\n[on screen right now: {title}]"
|
||||
|
||||
|
||||
# ── platform probes ──────────────────────────────────────────────────────────
|
||||
|
||||
def _linux_active_window_id() -> Optional[str]:
|
||||
if not shutil.which("xprop"):
|
||||
return None
|
||||
output = _run(["xprop", "-root", "_NET_ACTIVE_WINDOW"])
|
||||
return parse_xprop_window_id(output or "")
|
||||
|
||||
|
||||
def _linux_title() -> Optional[str]:
|
||||
if shutil.which("xdotool"):
|
||||
output = _run(["xdotool", "getactivewindow", "getwindowname"])
|
||||
if output and output.strip():
|
||||
return output.strip()
|
||||
window_id = _linux_active_window_id()
|
||||
if window_id is None:
|
||||
return None
|
||||
for prop in ("_NET_WM_NAME", "WM_NAME"):
|
||||
output = _run(["xprop", "-id", window_id, prop])
|
||||
title = parse_xprop_window_name(output or "")
|
||||
if title:
|
||||
return title
|
||||
return None
|
||||
|
||||
|
||||
def _windows_title() -> Optional[str]:
|
||||
try:
|
||||
import ctypes
|
||||
|
||||
user32 = ctypes.windll.user32
|
||||
handle = user32.GetForegroundWindow()
|
||||
if not handle:
|
||||
return None
|
||||
length = user32.GetWindowTextLengthW(handle)
|
||||
buffer = ctypes.create_unicode_buffer(length + 1)
|
||||
user32.GetWindowTextW(handle, buffer, length + 1)
|
||||
return buffer.value or None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _macos_title() -> Optional[str]:
|
||||
script = (
|
||||
'tell application "System Events" to get name of first application process '
|
||||
"whose frontmost is true"
|
||||
)
|
||||
output = _run(["osascript", "-e", script])
|
||||
return (output or "").strip() or None
|
||||
|
||||
|
||||
def active_window_title() -> Optional[str]:
|
||||
"""Focused window's title, or None if the platform won't tell us."""
|
||||
try:
|
||||
system = platform.system()
|
||||
if system == "Linux":
|
||||
return clean_title(_linux_title())
|
||||
if system == "Windows":
|
||||
return clean_title(_windows_title())
|
||||
if system == "Darwin":
|
||||
return clean_title(_macos_title())
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def is_fullscreen_active() -> bool:
|
||||
"""True when the focused window is fullscreen (call, game, presentation).
|
||||
False whenever we can't tell — do-not-disturb should be something you opt
|
||||
into, not something a failed probe turns on."""
|
||||
try:
|
||||
system = platform.system()
|
||||
if system == "Linux":
|
||||
window_id = _linux_active_window_id()
|
||||
if window_id is None:
|
||||
return False
|
||||
return parse_xprop_fullscreen(_run(["xprop", "-id", window_id, "_NET_WM_STATE"]) or "")
|
||||
if system == "Windows":
|
||||
import ctypes
|
||||
from ctypes import wintypes
|
||||
|
||||
user32 = ctypes.windll.user32
|
||||
handle = user32.GetForegroundWindow()
|
||||
if not handle:
|
||||
return False
|
||||
rect = wintypes.RECT()
|
||||
user32.GetWindowRect(handle, ctypes.byref(rect))
|
||||
screen_w = user32.GetSystemMetrics(0)
|
||||
screen_h = user32.GetSystemMetrics(1)
|
||||
return (rect.right - rect.left) >= screen_w and (rect.bottom - rect.top) >= screen_h
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def context_for(text: str) -> str:
|
||||
"""What the controller sends: *text* plus the active window title, when
|
||||
SCREEN_CONTEXT is on and there's a title worth mentioning."""
|
||||
from . import config
|
||||
|
||||
if not config.SCREEN_CONTEXT:
|
||||
return text
|
||||
return annotate(text, active_window_title())
|
||||
@@ -0,0 +1,125 @@
|
||||
"""HTTP client for Bolt's desk API (ai/desk_api.py on the server).
|
||||
|
||||
Protocol is identical to desk_client/bolt_desk.py in the main tmn-api repo —
|
||||
this pet is just another desk client, so it gets the exact same brain,
|
||||
memory, tools, and persona as Discord chat and the Linux voice client:
|
||||
|
||||
text -> POST /desk/converse
|
||||
[server may relay a shell command back to run on THIS machine]
|
||||
... -> POST /desk/tool_result (repeat until the server sends a reply)
|
||||
reply <- returned to caller
|
||||
|
||||
Kept dependency-free beyond `requests` so it's easy to unit test with mocks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Callable, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from . import config
|
||||
|
||||
_MAX_RELAY_HOPS = 16
|
||||
|
||||
|
||||
class ServerError(Exception):
|
||||
"""Raised when the server responds with an error payload or unreachable."""
|
||||
|
||||
|
||||
def _headers() -> dict:
|
||||
return {"X-Desk-Api-Key": config.API_KEY}
|
||||
|
||||
|
||||
def check_health(timeout: float = 10.0) -> dict:
|
||||
response = requests.get(f"{config.SERVER_URL}/desk/health", headers=_headers(), timeout=timeout)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
|
||||
|
||||
def run_local_command(command: str, timeout: int = None) -> str:
|
||||
"""Execute a command relayed by the server, exactly as bolt_desk.py does —
|
||||
"full desktop control" for things like "open firefox" or "how full is my
|
||||
disk". Runs as the current desktop user. See README security notes."""
|
||||
timeout = timeout or config.COMMAND_TIMEOUT_SECONDS
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command, shell=True, capture_output=True, text=True,
|
||||
timeout=timeout, cwd=str(Path.home()),
|
||||
)
|
||||
output = (completed.stdout or "") + (completed.stderr or "")
|
||||
return f"[exit {completed.returncode}]\n{output}"[:6000]
|
||||
except subprocess.TimeoutExpired:
|
||||
return f"[command timed out after {timeout}s]"
|
||||
except Exception as exc:
|
||||
return f"[command failed: {exc}]"
|
||||
|
||||
|
||||
def converse(
|
||||
text: str,
|
||||
on_command: Callable[[str], str] = run_local_command,
|
||||
timeout: float = 120.0,
|
||||
) -> str:
|
||||
"""Send one turn of conversation to the desk API, relaying any commands
|
||||
the server sends back until it produces a final reply.
|
||||
|
||||
*on_command* is injectable for tests; defaults to actually running the
|
||||
command locally (matching bolt_desk.py's behavior).
|
||||
"""
|
||||
headers = _headers()
|
||||
try:
|
||||
response = requests.post(
|
||||
f"{config.SERVER_URL}/desk/converse",
|
||||
json={"session_id": config.SESSION_ID, "text": text},
|
||||
headers=headers, timeout=timeout,
|
||||
)
|
||||
payload = response.json()
|
||||
except Exception as exc:
|
||||
raise ServerError(f"couldn't reach the server: {exc}") from exc
|
||||
|
||||
for _ in range(_MAX_RELAY_HOPS):
|
||||
if payload.get("type") != "command":
|
||||
break
|
||||
output = on_command(str(payload.get("command") or ""))
|
||||
try:
|
||||
response = requests.post(
|
||||
f"{config.SERVER_URL}/desk/tool_result",
|
||||
json={
|
||||
"session_id": config.SESSION_ID,
|
||||
"token": payload.get("token"),
|
||||
"output": output,
|
||||
},
|
||||
headers=headers, timeout=180,
|
||||
)
|
||||
payload = response.json()
|
||||
except Exception as exc:
|
||||
raise ServerError(f"couldn't reach the server during tool relay: {exc}") from exc
|
||||
|
||||
if payload.get("type") == "reply":
|
||||
return str(payload.get("text") or "")
|
||||
raise ServerError(str(payload.get("error") or "unknown server response"))
|
||||
|
||||
|
||||
def report_status(timeout: float = 15.0) -> Optional[str]:
|
||||
"""Heartbeat — lets the desk API attach a pending spoken announcement
|
||||
(proactive nudges, reminders fired since the last heartbeat) that the pet
|
||||
can speak unprompted, exactly like the phone/desk clients. A pet has no
|
||||
battery/GPS to report, so the device-status fields
|
||||
(battery/is_charging/latitude/longitude/address, all optional server-side)
|
||||
are simply omitted.
|
||||
|
||||
Returns the announcement text to speak, or None if there's nothing pending.
|
||||
"""
|
||||
try:
|
||||
response = requests.post(
|
||||
f"{config.SERVER_URL}/desk/report_status",
|
||||
json={"session_id": config.SESSION_ID}, headers=_headers(), timeout=timeout,
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
except Exception as exc:
|
||||
raise ServerError(f"heartbeat failed: {exc}") from exc
|
||||
reply = data.get("reply")
|
||||
return str(reply) if reply else None
|
||||
@@ -0,0 +1,130 @@
|
||||
"""Turn a server reply into something worth *hearing*.
|
||||
|
||||
The server's persona writes for a chat window: markdown emphasis, bullet
|
||||
lists, emoji, bare URLs. A TTS voice reads those literally ("asterisk
|
||||
asterisk OS colon", "https colon slash slash..."), so everything spoken goes
|
||||
through for_speech() first. Pure string logic, no Qt/audio imports — cheap to
|
||||
unit test (see tests/test_speech_text.py).
|
||||
|
||||
for_display() is the lighter sibling used for the speech bubble: it drops the
|
||||
markdown *syntax* but keeps emoji and punctuation, since those render fine.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
# Pictographs, symbols, flags, dingbats, arrows, box drawing, variation
|
||||
# selectors, ZWJ — anything a voice would either skip or read as a name
|
||||
# ("black right-pointing triangle").
|
||||
_EMOJI = re.compile(
|
||||
"["
|
||||
"\U0001F000-\U0001FAFF" # emoji / pictographs / symbols blocks
|
||||
"\U00002190-\U000021FF" # arrows
|
||||
"\U00002300-\U000023FF" # misc technical (⌘ ⏱ …)
|
||||
"\U000025A0-\U000027BF" # geometric shapes, misc symbols, dingbats
|
||||
"\U00002B00-\U00002BFF" # extra arrows / shapes
|
||||
"\U0000FE00-\U0000FE0F" # variation selectors
|
||||
"\U0001F1E6-\U0001F1FF" # regional indicators (flags)
|
||||
"\U0000200D" # zero-width joiner
|
||||
"]+",
|
||||
flags=re.UNICODE,
|
||||
)
|
||||
|
||||
# Characters that are markup or decoration rather than speech. Kept out of
|
||||
# the spoken text entirely; ordinary punctuation (. , ! ? ; : ' " ( ) -) is
|
||||
# preserved because it shapes prosody.
|
||||
_UNSPEAKABLE = re.compile(r"[*_#`~^|<>\\{}\[\]/=+@©®™•·–—]+")
|
||||
|
||||
_FENCED_CODE = re.compile(r"```.*?```", re.DOTALL)
|
||||
_INLINE_CODE = re.compile(r"`([^`]*)`")
|
||||
_MD_IMAGE = re.compile(r"!\[([^\]]*)\]\([^)]*\)")
|
||||
_MD_LINK = re.compile(r"\[([^\]]+)\]\([^)]*\)")
|
||||
_URL = re.compile(r"\b(?:https?://|www\.)\S+")
|
||||
_HEADING = re.compile(r"^\s{0,3}#{1,6}\s*", re.MULTILINE)
|
||||
_BLOCKQUOTE = re.compile(r"^\s{0,3}>\s?", re.MULTILINE)
|
||||
_RULE = re.compile(r"^\s*([-*_=])(?:\s*\1){2,}\s*$", re.MULTILINE)
|
||||
_BULLET = re.compile(r"^\s*(?:[-*+•·]|\d+[.)])\s+", re.MULTILINE)
|
||||
_EMPHASIS = re.compile(r"(\*{1,3}|_{1,3})(\S(?:.*?\S)?)\1", re.DOTALL)
|
||||
_TABLE_PIPE = re.compile(r"[ \t]*\|[ \t]*")
|
||||
|
||||
# Symbols worth saying out loud rather than dropping — a bare "&" read as
|
||||
# nothing turns "R&D" into "RD". Split in two because the non-ASCII ones sit
|
||||
# inside the arrow/symbol blocks _EMOJI strips, so they have to be worded
|
||||
# before that pass; the ASCII ones must wait until *after* markdown parsing
|
||||
# (an "=" turned into " equals " would stop a "====" rule matching _RULE).
|
||||
_PRE_SPOKEN_SYMBOLS = {
|
||||
"→": " to ",
|
||||
"×": " by ",
|
||||
"°": " degrees ",
|
||||
"≈": " about ",
|
||||
}
|
||||
_SPOKEN_SYMBOLS = {
|
||||
"&": " and ",
|
||||
"%": " percent ",
|
||||
"@": " at ",
|
||||
"+": " plus ",
|
||||
"=": " equals ",
|
||||
}
|
||||
|
||||
_MULTI_SPACE = re.compile(r"[ \t]+")
|
||||
_MULTI_PUNCT = re.compile(r"(?:\s*\.){2,}")
|
||||
|
||||
|
||||
def _strip_markdown(text: str, *, keep_emoji: bool) -> str:
|
||||
text = _FENCED_CODE.sub(" (code) ", text)
|
||||
text = _INLINE_CODE.sub(r"\1", text)
|
||||
text = _MD_IMAGE.sub(r"\1", text)
|
||||
text = _MD_LINK.sub(r"\1", text)
|
||||
text = _RULE.sub("", text)
|
||||
text = _HEADING.sub("", text)
|
||||
text = _BLOCKQUOTE.sub("", text)
|
||||
text = _EMPHASIS.sub(r"\2", text)
|
||||
if not keep_emoji:
|
||||
text = _EMOJI.sub(" ", text)
|
||||
return text
|
||||
|
||||
|
||||
def _bullets_to_sentences(text: str) -> str:
|
||||
"""A read-aloud list needs pauses where the eye would see line breaks,
|
||||
otherwise "OS Linux Uptime 1 day CPU load moderate" runs together."""
|
||||
lines = [_BULLET.sub("", line).strip() for line in text.splitlines()]
|
||||
lines = [line for line in lines if line]
|
||||
if len(lines) < 2:
|
||||
return lines[0] if lines else ""
|
||||
# A bullet like "OS: Linux" ends mid-thought — give the voice a full stop
|
||||
# so consecutive items don't slur into one run-on sentence.
|
||||
return " ".join(line if line[-1] in ".!?:,;" else line + "." for line in lines)
|
||||
|
||||
|
||||
def for_speech(text: str) -> str:
|
||||
"""Plain prose for the TTS engine: no markdown, no emoji, no bare URLs,
|
||||
no stray symbols that would be read out character by character."""
|
||||
text = (text or "").strip()
|
||||
if not text:
|
||||
return ""
|
||||
for symbol, spoken in _PRE_SPOKEN_SYMBOLS.items():
|
||||
text = text.replace(symbol, spoken)
|
||||
text = _strip_markdown(text, keep_emoji=False)
|
||||
text = _URL.sub(" link ", text)
|
||||
text = _TABLE_PIPE.sub(", ", text)
|
||||
text = _bullets_to_sentences(text)
|
||||
for symbol, spoken in _SPOKEN_SYMBOLS.items():
|
||||
text = text.replace(symbol, spoken)
|
||||
text = _UNSPEAKABLE.sub(" ", text)
|
||||
text = _MULTI_PUNCT.sub(".", text)
|
||||
text = _MULTI_SPACE.sub(" ", text)
|
||||
text = re.sub(r"\s+([.,!?;:])", r"\1", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def for_display(text: str) -> str:
|
||||
"""What the speech bubble shows: markdown syntax removed (the bubble
|
||||
can't render it) but emoji and layout-ish punctuation left alone."""
|
||||
text = (text or "").strip()
|
||||
if not text:
|
||||
return ""
|
||||
text = _strip_markdown(text, keep_emoji=True)
|
||||
lines = [_BULLET.sub("• ", line).strip() for line in text.splitlines()]
|
||||
text = " ".join(line for line in lines if line)
|
||||
return _MULTI_SPACE.sub(" ", text).strip()
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Pet state machine — pure logic, no Qt/audio dependencies, so it's cheap
|
||||
to unit test. The UI layer (ui/pet_window.py) reacts to state changes by
|
||||
swapping the active sprite animation; the worker thread (ui/app.py) drives
|
||||
transitions as the mic/wake/converse/tts pipeline progresses.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
from typing import Callable, Optional
|
||||
|
||||
|
||||
class PetState(str, Enum):
|
||||
IDLE = "idle" # waiting for the wake phrase (or a click)
|
||||
LISTENING = "listening" # actively recording an utterance
|
||||
THINKING = "thinking" # waiting on the server (STT done, converse in flight)
|
||||
TALKING = "talking" # playing back the TTS reply
|
||||
ERROR = "error" # brief flash state on failure, then back to idle
|
||||
|
||||
|
||||
# States it's valid to move to from each state. Keeps ad-hoc bugs (e.g.
|
||||
# firing TALKING before a reply exists) from silently passing through.
|
||||
#
|
||||
# IDLE -> TALKING is legal (not just IDLE -> LISTENING) because of proactive
|
||||
# announcements: the heartbeat poll (controller.py's _maybe_heartbeat) can
|
||||
# make the pet speak unprompted — a reminder firing, a nudge from the server
|
||||
# — without the user having said anything first, so there's no preceding
|
||||
# LISTENING/THINKING leg for that turn.
|
||||
_TRANSITIONS: dict[PetState, set[PetState]] = {
|
||||
PetState.IDLE: {PetState.LISTENING, PetState.TALKING, PetState.ERROR},
|
||||
PetState.LISTENING: {PetState.THINKING, PetState.IDLE, PetState.ERROR},
|
||||
PetState.THINKING: {PetState.TALKING, PetState.IDLE, PetState.ERROR},
|
||||
PetState.TALKING: {PetState.IDLE, PetState.ERROR},
|
||||
PetState.ERROR: {PetState.IDLE},
|
||||
}
|
||||
|
||||
|
||||
class InvalidTransition(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class PetStateMachine:
|
||||
def __init__(self, on_change: Optional[Callable[[PetState, PetState], None]] = None):
|
||||
self._state = PetState.IDLE
|
||||
self._on_change = on_change
|
||||
|
||||
@property
|
||||
def state(self) -> PetState:
|
||||
return self._state
|
||||
|
||||
def transition(self, new_state: PetState) -> None:
|
||||
if new_state == self._state:
|
||||
return
|
||||
allowed = _TRANSITIONS.get(self._state, set())
|
||||
if new_state not in allowed:
|
||||
raise InvalidTransition(f"{self._state} -> {new_state} is not allowed")
|
||||
old_state = self._state
|
||||
self._state = new_state
|
||||
if self._on_change is not None:
|
||||
self._on_change(old_state, new_state)
|
||||
|
||||
def force(self, new_state: PetState) -> None:
|
||||
"""Bypass the transition table — used only for recovering to IDLE
|
||||
from an unexpected/edge-case state (e.g. after an exception mid
|
||||
pipeline). Prefer transition() everywhere else."""
|
||||
old_state = self._state
|
||||
if new_state == old_state:
|
||||
return
|
||||
self._state = new_state
|
||||
if self._on_change is not None:
|
||||
self._on_change(old_state, new_state)
|
||||
@@ -0,0 +1,103 @@
|
||||
"""Wires everything together: QApplication, the pet window, the tray icon,
|
||||
the history / wake-tuner windows, the global push-to-talk hotkey, and the
|
||||
background PetController thread that owns the mic/wake/server/TTS pipeline.
|
||||
|
||||
Everything the controller wants the UI to do arrives as a Qt signal, so the
|
||||
worker thread never touches a widget.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
from PySide6.QtCore import QThread
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
from .. import config
|
||||
from ..controller import PetController
|
||||
from ..hotkey import GlobalHotkey
|
||||
from ..state import PetState
|
||||
from .history_window import HistoryWindow
|
||||
from .pet_window import PetWindow
|
||||
from .tray import PetTray
|
||||
from .wake_tuner import WakeTunerWindow
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(message, flush=True)
|
||||
|
||||
|
||||
def run() -> int:
|
||||
app = QApplication(sys.argv)
|
||||
app.setQuitOnLastWindowClosed(False) # tray-driven app; closing the pet isn't "quit"
|
||||
|
||||
window = PetWindow()
|
||||
window.show()
|
||||
|
||||
controller = PetController()
|
||||
thread = QThread()
|
||||
controller.moveToThread(thread)
|
||||
|
||||
thread.started.connect(controller.run)
|
||||
controller.state_changed.connect(lambda value: window.set_state(PetState(value)))
|
||||
controller.said.connect(window.say)
|
||||
controller.log.connect(_log)
|
||||
controller.action.connect(window.apply_action) # petctl move/emote/say/...
|
||||
controller.finished.connect(thread.quit)
|
||||
|
||||
window.talk_requested.connect(controller.request_talk_now)
|
||||
window.copied.connect(lambda text: _log(f"Copied to clipboard: {text[:60]}"))
|
||||
|
||||
history_window = HistoryWindow(controller.history)
|
||||
tuner_window = WakeTunerWindow(
|
||||
get_threshold=controller.wake_threshold,
|
||||
set_threshold=controller.set_wake_threshold,
|
||||
get_stats=controller.wake_stats,
|
||||
on_reset=controller.reset_wake_stats,
|
||||
)
|
||||
|
||||
def _set_nap(napping: bool) -> None:
|
||||
# Clicking the tray item pins the state; the schedule takes over again
|
||||
# only after a restart or a `petctl nap off`.
|
||||
controller.set_napping(napping)
|
||||
window.set_napping(napping)
|
||||
|
||||
tray = PetTray(
|
||||
on_talk_now=controller.request_talk_now,
|
||||
on_toggle_mute=controller.toggle_mute,
|
||||
on_quit=app.quit,
|
||||
on_set_wander=window.set_wander_enabled,
|
||||
wander_enabled=config.PET_WANDER,
|
||||
on_set_click_through=window.set_click_through,
|
||||
click_through_enabled=config.PET_CLICK_THROUGH,
|
||||
on_set_nap=_set_nap,
|
||||
on_show_history=history_window.show_refreshed,
|
||||
on_show_wake_tuner=tuner_window.show_refreshed,
|
||||
)
|
||||
|
||||
def _handle_napping(napping: bool) -> None:
|
||||
window.set_napping(napping)
|
||||
tray.set_napping(napping)
|
||||
|
||||
controller.napping.connect(_handle_napping)
|
||||
|
||||
# Push-to-talk: a global hook, because the pet window never has focus.
|
||||
# request_talk_now() only sets a threading.Event, so it's safe to call
|
||||
# from pynput's listener thread.
|
||||
hotkey = GlobalHotkey(config.PUSH_TO_TALK_HOTKEY, controller.request_talk_now)
|
||||
problem = hotkey.start()
|
||||
if problem:
|
||||
_log(problem)
|
||||
elif hotkey.running:
|
||||
_log(f"Push-to-talk: {config.PUSH_TO_TALK_HOTKEY}")
|
||||
|
||||
def _shutdown() -> None:
|
||||
hotkey.stop()
|
||||
controller.stop()
|
||||
thread.quit()
|
||||
thread.wait(5000)
|
||||
|
||||
app.aboutToQuit.connect(_shutdown)
|
||||
|
||||
thread.start()
|
||||
return app.exec()
|
||||
@@ -0,0 +1,83 @@
|
||||
"""Scrollback for the speech bubble.
|
||||
|
||||
The bubble is transient by design, so anything Bolt said more than a few
|
||||
seconds ago is gone. This is the "wait, what was that path again?" window:
|
||||
the last HISTORY_LIMIT turns, copyable. Opened from the tray.
|
||||
|
||||
Reads a bolt_pet.history.ConversationHistory (pure logic, tested separately);
|
||||
this file is only presentation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from typing import Callable, Optional
|
||||
|
||||
from PySide6.QtCore import Qt
|
||||
from PySide6.QtWidgets import (
|
||||
QApplication, QDialog, QHBoxLayout, QPlainTextEdit, QPushButton, QVBoxLayout,
|
||||
)
|
||||
|
||||
from ..history import ConversationHistory
|
||||
|
||||
|
||||
def _clock(timestamp: float) -> str:
|
||||
return time.strftime("%H:%M:%S", time.localtime(timestamp))
|
||||
|
||||
|
||||
class HistoryWindow(QDialog):
|
||||
def __init__(self, history: ConversationHistory, on_clear: Optional[Callable[[], None]] = None):
|
||||
super().__init__()
|
||||
self._history = history
|
||||
self._on_clear = on_clear
|
||||
self.setWindowTitle("Bolt — conversation history")
|
||||
self.resize(620, 420)
|
||||
|
||||
self._view = QPlainTextEdit()
|
||||
self._view.setReadOnly(True)
|
||||
self._view.setLineWrapMode(QPlainTextEdit.WidgetWidth)
|
||||
|
||||
copy_button = QPushButton("Copy all")
|
||||
copy_button.clicked.connect(self._copy_all)
|
||||
clear_button = QPushButton("Clear")
|
||||
clear_button.clicked.connect(self._clear)
|
||||
close_button = QPushButton("Close")
|
||||
close_button.clicked.connect(self.close)
|
||||
close_button.setDefault(True)
|
||||
|
||||
buttons = QHBoxLayout()
|
||||
buttons.addWidget(copy_button)
|
||||
buttons.addWidget(clear_button)
|
||||
buttons.addStretch(1)
|
||||
buttons.addWidget(close_button)
|
||||
|
||||
layout = QVBoxLayout(self)
|
||||
layout.addWidget(self._view)
|
||||
layout.addLayout(buttons)
|
||||
|
||||
def refresh(self) -> None:
|
||||
self._view.setPlainText(self._history.as_text(clock=_clock))
|
||||
# Jump to the newest line — that's what you opened this for.
|
||||
scrollbar = self._view.verticalScrollBar()
|
||||
scrollbar.setValue(scrollbar.maximum())
|
||||
|
||||
def show_refreshed(self) -> None:
|
||||
self.refresh()
|
||||
self.show()
|
||||
self.raise_()
|
||||
self.activateWindow()
|
||||
|
||||
def _copy_all(self) -> None:
|
||||
QApplication.clipboard().setText(self._history.as_text(clock=_clock))
|
||||
|
||||
def _clear(self) -> None:
|
||||
self._history.clear()
|
||||
if self._on_clear is not None:
|
||||
self._on_clear()
|
||||
self.refresh()
|
||||
|
||||
def keyPressEvent(self, event) -> None:
|
||||
if event.key() == Qt.Key_Escape:
|
||||
self.close()
|
||||
return
|
||||
super().keyPressEvent(event)
|
||||
@@ -0,0 +1,582 @@
|
||||
"""The pet itself: a frameless, translucent, always-on-top window that
|
||||
renders the current sprite animation, wanders the desktop on its own while
|
||||
idle, walks/emotes on command from the server (see pet_actions.py), can be
|
||||
dragged around, dims when napping, and turns a plain (non-drag) click into a
|
||||
"talk now" request.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import random
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from PySide6.QtCore import QPoint, Qt, QTimer, Signal
|
||||
from PySide6.QtGui import (
|
||||
QColor, QCursor, QFont, QFontMetrics, QPainter, QPainterPath, QPixmap, QRegion, QTransform,
|
||||
)
|
||||
from PySide6.QtWidgets import QApplication, QWidget
|
||||
|
||||
from .. import config
|
||||
from ..state import PetState
|
||||
from .sprite import SpriteSet
|
||||
|
||||
_DRAG_THRESHOLD_PX = 4
|
||||
# Movement runs on its own ~30fps timer, independent of the (slower) sprite
|
||||
# animation timer, so a stroll looks smooth even at IDLE_ANIMATION_FPS=6.
|
||||
_WANDER_TICK_MS = 33
|
||||
_EMOTE_TICKS = 36 # ~1.2s per emote at the tick rate above
|
||||
_NAP_OPACITY = 0.35
|
||||
|
||||
|
||||
def emote_transform(emote: str, progress: float) -> tuple[float, float, float, float]:
|
||||
"""(dx, dy, rotation_degrees, scale) for an emote at *progress* 0..1.
|
||||
|
||||
Pure maths, deliberately separate from paintEvent so the motion curves can
|
||||
be unit tested (and so adding an emote doesn't mean touching painting
|
||||
code). Every emote must return to (0, 0, 0, 1) at progress 1.0, otherwise
|
||||
the pet ends up permanently askew.
|
||||
"""
|
||||
progress = min(max(progress, 0.0), 1.0)
|
||||
fade = math.sin(math.pi * progress) # 0 -> 1 -> 0, so it always lands home
|
||||
tau = 2 * math.pi
|
||||
if emote == "wave":
|
||||
return 0.0, 0.0, 14.0 * fade * math.sin(tau * 2 * progress), 1.0
|
||||
if emote in ("hop", "bounce"):
|
||||
hops = 2 if emote == "hop" else 3
|
||||
return 0.0, -22.0 * fade * abs(math.sin(math.pi * hops * progress)), 0.0, 1.0
|
||||
if emote == "spin":
|
||||
return 0.0, 0.0, 360.0 * progress % 360.0, 1.0
|
||||
if emote == "nod":
|
||||
return 0.0, 10.0 * fade * math.sin(tau * 2 * progress), 0.0, 1.0 - 0.05 * fade
|
||||
if emote in ("shake", "wiggle"):
|
||||
return 14.0 * fade * math.sin(tau * 3 * progress), 0.0, 0.0, 1.0
|
||||
return 0.0, 0.0, 0.0, 1.0
|
||||
|
||||
|
||||
class SpeechBubble(QWidget):
|
||||
"""Small translucent word-bubble shown above the pet while it talks."""
|
||||
|
||||
_MAX_WIDTH = 260
|
||||
_PADDING = 10
|
||||
|
||||
copied = Signal(str)
|
||||
|
||||
def __init__(self, parent: Optional[QWidget] = None):
|
||||
super().__init__(parent, Qt.FramelessWindowHint | Qt.Tool)
|
||||
self.setAttribute(Qt.WA_TranslucentBackground)
|
||||
self.setAttribute(Qt.WA_ShowWithoutActivating)
|
||||
self._text = ""
|
||||
self._font = QFont()
|
||||
self._font.setPointSize(10)
|
||||
self._flash = "" # transient overlay ("Copied") drawn over the text
|
||||
self._hide_timer = QTimer(self)
|
||||
self._hide_timer.setSingleShot(True)
|
||||
self._hide_timer.timeout.connect(self.hide)
|
||||
self._flash_timer = QTimer(self)
|
||||
self._flash_timer.setSingleShot(True)
|
||||
self._flash_timer.timeout.connect(self._clear_flash)
|
||||
self.setToolTip("Click to copy")
|
||||
self.setCursor(Qt.PointingHandCursor)
|
||||
self.hide()
|
||||
|
||||
@property
|
||||
def text(self) -> str:
|
||||
return self._text
|
||||
|
||||
def show_text(self, text: str, duration_ms: int = 6000) -> None:
|
||||
text = (text or "").strip()
|
||||
if not text:
|
||||
self.hide()
|
||||
return
|
||||
self._text = text
|
||||
self._relayout()
|
||||
self.show()
|
||||
self.raise_()
|
||||
self._hide_timer.start(duration_ms)
|
||||
|
||||
# ── click to copy ────────────────────────────────────────────────────
|
||||
# The bubble hides itself after a few seconds, which is fine for chat and
|
||||
# awful for anything you needed to keep (a path, a number, a command's
|
||||
# output). One click puts it on the clipboard; the tray's History window
|
||||
# has the rest.
|
||||
|
||||
def mousePressEvent(self, event) -> None:
|
||||
if event.button() != Qt.LeftButton or not self._text:
|
||||
return
|
||||
QApplication.clipboard().setText(self._text)
|
||||
self.copied.emit(self._text)
|
||||
self._flash = "Copied to clipboard"
|
||||
self.update()
|
||||
self._flash_timer.start(900)
|
||||
self._hide_timer.start(2500) # linger a moment so the flash is visible
|
||||
|
||||
def _clear_flash(self) -> None:
|
||||
self._flash = ""
|
||||
self.update()
|
||||
|
||||
def _wrapped_lines(self) -> list[str]:
|
||||
metrics = QFontMetrics(self._font)
|
||||
words = self._text.split()
|
||||
lines: list[str] = []
|
||||
current = ""
|
||||
max_text_width = self._MAX_WIDTH - 2 * self._PADDING
|
||||
for word in words:
|
||||
candidate = f"{current} {word}".strip()
|
||||
if metrics.horizontalAdvance(candidate) <= max_text_width or not current:
|
||||
current = candidate
|
||||
else:
|
||||
lines.append(current)
|
||||
current = word
|
||||
if current:
|
||||
lines.append(current)
|
||||
return lines[:6] # don't let a huge reply turn into a wall of bubble
|
||||
|
||||
def _relayout(self) -> None:
|
||||
metrics = QFontMetrics(self._font)
|
||||
lines = self._wrapped_lines()
|
||||
text_width = max((metrics.horizontalAdvance(line) for line in lines), default=0)
|
||||
width = min(self._MAX_WIDTH, text_width + 2 * self._PADDING)
|
||||
height = metrics.height() * len(lines) + 2 * self._PADDING
|
||||
self.resize(width, height)
|
||||
|
||||
def paintEvent(self, _event) -> None:
|
||||
painter = QPainter(self)
|
||||
painter.setRenderHint(QPainter.Antialiasing)
|
||||
path = QPainterPath()
|
||||
path.addRoundedRect(0, 0, self.width(), self.height(), 10, 10)
|
||||
painter.fillPath(path, QColor(30, 30, 35, 230))
|
||||
painter.setFont(self._font)
|
||||
metrics = QFontMetrics(self._font)
|
||||
if self._flash:
|
||||
painter.setPen(QColor(150, 230, 170))
|
||||
painter.drawText(self.rect(), Qt.AlignCenter, self._flash)
|
||||
return
|
||||
painter.setPen(QColor(240, 240, 245))
|
||||
y = self._PADDING + metrics.ascent()
|
||||
for line in self._wrapped_lines():
|
||||
painter.drawText(self._PADDING, y, line)
|
||||
y += metrics.height()
|
||||
|
||||
|
||||
class PetWindow(QWidget):
|
||||
talk_requested = Signal()
|
||||
copied = Signal(str) # bubble text the user just put on the clipboard
|
||||
|
||||
def __init__(self, sprite_dir: Optional[Path] = None, size: Optional[int] = None):
|
||||
super().__init__()
|
||||
flags = Qt.FramelessWindowHint | Qt.Tool
|
||||
if config.PET_ALWAYS_ON_TOP:
|
||||
flags |= Qt.WindowStaysOnTopHint
|
||||
self.setWindowFlags(flags)
|
||||
self.setAttribute(Qt.WA_TranslucentBackground)
|
||||
|
||||
self.sprites = SpriteSet(sprite_dir or (Path(__file__).resolve().parent.parent / "assets" / "sprites"),
|
||||
size or config.PET_SIZE)
|
||||
self.resize(self.sprites.size, self.sprites.size)
|
||||
self._current_state = PetState.IDLE
|
||||
|
||||
self._drag_offset: Optional[QPoint] = None
|
||||
self._press_pos: Optional[QPoint] = None
|
||||
self._dragged = False
|
||||
|
||||
self._bubble = SpeechBubble()
|
||||
self._bubble.copied.connect(self.copied)
|
||||
|
||||
self._napping = False
|
||||
self._emote: Optional[str] = None
|
||||
self._emote_tick = 0
|
||||
self._mask_key = None
|
||||
|
||||
self._anim_timer = QTimer(self)
|
||||
self._anim_timer.timeout.connect(self._advance_frame)
|
||||
fps = max(1.0, config.IDLE_ANIMATION_FPS)
|
||||
self._anim_timer.start(int(1000 / fps))
|
||||
|
||||
self._wander_enabled = config.PET_WANDER
|
||||
self._wander_target: Optional[QPoint] = None
|
||||
self._commanded_move = False # a petctl move — happens even mid-conversation
|
||||
self._next_wander_at = 0.0
|
||||
self._bob_offset = 0
|
||||
self._bob_phase = 0.0
|
||||
self._schedule_next_wander()
|
||||
self._wander_timer = QTimer(self)
|
||||
self._wander_timer.timeout.connect(self._movement_tick)
|
||||
self._wander_timer.start(_WANDER_TICK_MS)
|
||||
|
||||
self._click_through = False
|
||||
self.set_click_through(config.PET_CLICK_THROUGH)
|
||||
self._place_start_position()
|
||||
|
||||
# ── placement ────────────────────────────────────────────────────────
|
||||
|
||||
def _place_start_position(self) -> None:
|
||||
screen = QApplication.primaryScreen()
|
||||
geo = screen.availableGeometry() if screen else None
|
||||
try:
|
||||
x = int(config.PET_START_X) if config.PET_START_X else None
|
||||
y = int(config.PET_START_Y) if config.PET_START_Y else None
|
||||
except ValueError:
|
||||
x = y = None
|
||||
if geo is not None:
|
||||
x = geo.right() - self.width() - 40 if x is None else x
|
||||
y = geo.bottom() - self.height() - 60 if y is None else y
|
||||
self.move(x or 0, y or 0)
|
||||
self._reposition_bubble()
|
||||
|
||||
def _reposition_bubble(self) -> None:
|
||||
top_left = self.geometry().topLeft()
|
||||
self._bubble.move(
|
||||
top_left.x() + self.width() // 2 - self._bubble.width() // 2,
|
||||
top_left.y() - self._bubble.height() - 8,
|
||||
)
|
||||
|
||||
# ── server-driven actions (petctl) ───────────────────────────────────
|
||||
|
||||
def apply_action(self, action: dict) -> None:
|
||||
"""Perform one parsed petctl action (see pet_actions.py). Called on
|
||||
the UI thread via a queued signal from the controller."""
|
||||
kind = action.get("action")
|
||||
if kind == "move":
|
||||
target = self._resolve_move_target(action)
|
||||
if target is not None:
|
||||
self._wander_target = target
|
||||
self._commanded_move = True # overrides the idle-only rule
|
||||
elif kind == "emote":
|
||||
self.start_emote(action["emote"])
|
||||
elif kind == "say":
|
||||
self.say(action["text"])
|
||||
elif kind == "wander":
|
||||
self.set_wander_enabled(bool(action["enabled"]))
|
||||
elif kind == "nap":
|
||||
self.set_napping(bool(action["enabled"]))
|
||||
|
||||
def _resolve_move_target(self, action: dict) -> Optional[QPoint]:
|
||||
geo = self._screen_geometry()
|
||||
if "x" in action and "y" in action:
|
||||
point = QPoint(int(action["x"]), int(action["y"]))
|
||||
return self._clamp_to_screen(point, geo)
|
||||
anchor = action.get("anchor")
|
||||
if anchor == "cursor":
|
||||
cursor = QCursor.pos()
|
||||
return self._clamp_to_screen(
|
||||
QPoint(cursor.x() - self.width() // 2, cursor.y() - self.height() // 2), geo
|
||||
)
|
||||
if geo is None:
|
||||
return None
|
||||
if anchor == "random":
|
||||
return self._pick_wander_target()
|
||||
margin = config.PET_WANDER_MARGIN
|
||||
left, right = geo.left() + margin, geo.right() - self.width() - margin
|
||||
top, bottom = geo.top() + margin, geo.bottom() - self.height() - margin
|
||||
middle_x = geo.left() + (geo.width() - self.width()) // 2
|
||||
middle_y = geo.top() + (geo.height() - self.height()) // 2
|
||||
positions = {
|
||||
"top-left": (left, top), "top": (middle_x, top), "top-right": (right, top),
|
||||
"left": (left, middle_y), "center": (middle_x, middle_y), "right": (right, middle_y),
|
||||
"bottom-left": (left, bottom), "bottom": (middle_x, bottom), "bottom-right": (right, bottom),
|
||||
}
|
||||
if anchor not in positions:
|
||||
return None
|
||||
return QPoint(*positions[anchor])
|
||||
|
||||
def _clamp_to_screen(self, point: QPoint, geo) -> QPoint:
|
||||
if geo is None:
|
||||
return point
|
||||
x = min(max(point.x(), geo.left()), max(geo.left(), geo.right() - self.width()))
|
||||
y = min(max(point.y(), geo.top()), max(geo.top(), geo.bottom() - self.height()))
|
||||
return QPoint(x, y)
|
||||
|
||||
# ── emotes ───────────────────────────────────────────────────────────
|
||||
|
||||
def start_emote(self, emote: str) -> None:
|
||||
self._emote = emote
|
||||
self._emote_tick = 0
|
||||
self.update()
|
||||
|
||||
def _advance_emote(self) -> None:
|
||||
if self._emote is None:
|
||||
return
|
||||
self._emote_tick += 1
|
||||
if self._emote_tick > _EMOTE_TICKS:
|
||||
self._emote = None
|
||||
self._emote_tick = 0
|
||||
self.update()
|
||||
|
||||
def _emote_transform(self) -> tuple[float, float, float, float]:
|
||||
if self._emote is None:
|
||||
return 0.0, 0.0, 0.0, 1.0
|
||||
return emote_transform(self._emote, self._emote_tick / _EMOTE_TICKS)
|
||||
|
||||
# ── napping (quiet hours / do-not-disturb) ───────────────────────────
|
||||
|
||||
def set_napping(self, napping: bool) -> None:
|
||||
"""Dim and stand still. Purely cosmetic here — the controller is what
|
||||
actually suppresses proactive speech."""
|
||||
if napping == self._napping:
|
||||
return
|
||||
self._napping = napping
|
||||
self.setWindowOpacity(_NAP_OPACITY if napping else 1.0)
|
||||
if napping:
|
||||
self._stop_walking()
|
||||
self.update()
|
||||
|
||||
@property
|
||||
def napping(self) -> bool:
|
||||
return self._napping
|
||||
|
||||
# ── mouse transparency ───────────────────────────────────────────────
|
||||
|
||||
def set_click_through(self, enabled: bool) -> None:
|
||||
"""When on, the pet ignores the mouse entirely (tray-only control) —
|
||||
for when it's parked over something you need to click a lot."""
|
||||
self._click_through = enabled
|
||||
self.setAttribute(Qt.WA_TransparentForMouseEvents, enabled)
|
||||
if enabled:
|
||||
self.clearMask()
|
||||
self._mask_key = None
|
||||
else:
|
||||
self._mask_key = None # force the shaped mask to be rebuilt
|
||||
|
||||
@property
|
||||
def click_through(self) -> bool:
|
||||
return self._click_through
|
||||
|
||||
def _apply_input_mask(self, pixmap: Optional[QPixmap], x: int, y: int) -> None:
|
||||
"""Restrict the window to the sprite's opaque pixels, so the square
|
||||
window's transparent corners stop swallowing clicks meant for what's
|
||||
underneath. Rebuilt only when the frame actually changes — the mask
|
||||
is derived from the pixmap's alpha, which isn't free."""
|
||||
if self._click_through or not config.PET_SHAPED_INPUT:
|
||||
return
|
||||
if pixmap is None:
|
||||
if self._mask_key is not None:
|
||||
self.clearMask()
|
||||
self._mask_key = None
|
||||
return
|
||||
key = (pixmap.cacheKey(), x, y)
|
||||
if key == self._mask_key:
|
||||
return
|
||||
self._mask_key = key
|
||||
try:
|
||||
region = QRegion(pixmap.mask())
|
||||
region.translate(x, y)
|
||||
self.setMask(region)
|
||||
except Exception:
|
||||
self.clearMask() # a sprite without an alpha channel — never mind
|
||||
|
||||
# ── wandering ────────────────────────────────────────────────────────
|
||||
|
||||
def set_wander_enabled(self, enabled: bool) -> None:
|
||||
self._wander_enabled = enabled
|
||||
if not enabled:
|
||||
self._stop_walking()
|
||||
|
||||
def wander_now(self) -> None:
|
||||
"""Stroll immediately (tray menu / anything that wants a nudge)."""
|
||||
self._next_wander_at = 0.0
|
||||
|
||||
def _screen_geometry(self):
|
||||
# screenAt() so a multi-monitor setup keeps the pet on the screen
|
||||
# it's currently standing on rather than yanking it to the primary.
|
||||
screen = QApplication.screenAt(self.frameGeometry().center()) or QApplication.primaryScreen()
|
||||
return screen.availableGeometry() if screen else None
|
||||
|
||||
def _schedule_next_wander(self) -> None:
|
||||
base = max(1.0, config.PET_WANDER_INTERVAL_SECONDS)
|
||||
self._next_wander_at = time.monotonic() + random.uniform(0.5 * base, 1.5 * base)
|
||||
|
||||
def _stop_walking(self) -> None:
|
||||
if self._wander_target is None and not self._bob_offset:
|
||||
return
|
||||
self._wander_target = None
|
||||
self._commanded_move = False
|
||||
self._bob_phase = 0.0
|
||||
self._bob_offset = 0
|
||||
self.update()
|
||||
|
||||
def snap_to_edge(self) -> bool:
|
||||
"""If the pet has come to rest near a screen edge, tuck it flush
|
||||
against it — a desktop pet parked 11px off the taskbar looks like a
|
||||
bug. Returns True if it moved."""
|
||||
geo = self._screen_geometry()
|
||||
if geo is None or not config.PET_EDGE_SNAP:
|
||||
return False
|
||||
margin = config.PET_SNAP_MARGIN
|
||||
here = self.pos()
|
||||
x, y = here.x(), here.y()
|
||||
if abs(x - geo.left()) <= margin:
|
||||
x = geo.left()
|
||||
elif abs(geo.right() - (x + self.width())) <= margin:
|
||||
x = geo.right() - self.width() + 1
|
||||
if abs(y - geo.top()) <= margin:
|
||||
y = geo.top()
|
||||
elif abs(geo.bottom() - (y + self.height())) <= margin:
|
||||
y = geo.bottom() - self.height() + 1
|
||||
if (x, y) == (here.x(), here.y()):
|
||||
return False
|
||||
self.move(x, y)
|
||||
self._reposition_bubble()
|
||||
return True
|
||||
|
||||
def _pick_wander_target(self) -> Optional[QPoint]:
|
||||
geo = self._screen_geometry()
|
||||
if geo is None:
|
||||
return None
|
||||
margin = config.PET_WANDER_MARGIN
|
||||
min_x, max_x = geo.left() + margin, geo.right() - self.width() - margin
|
||||
min_y, max_y = geo.top() + margin, geo.bottom() - self.height() - margin
|
||||
if max_x <= min_x or max_y <= min_y: # pet bigger than the screen
|
||||
return None
|
||||
here = self.pos()
|
||||
target = QPoint(random.randint(min_x, max_x), random.randint(min_y, max_y))
|
||||
dx, dy = target.x() - here.x(), target.y() - here.y()
|
||||
distance = math.hypot(dx, dy)
|
||||
limit = max(1.0, config.PET_WANDER_MAX_DISTANCE)
|
||||
if distance > limit: # shorten the trip rather than sprinting the diagonal
|
||||
scale = limit / distance
|
||||
target = QPoint(round(here.x() + dx * scale), round(here.y() + dy * scale))
|
||||
elif distance < 8: # already there — not worth a stroll
|
||||
return None
|
||||
return target
|
||||
|
||||
def _movement_tick(self) -> None:
|
||||
"""One timer, two jobs — emotes play whatever the pet is doing, while
|
||||
wandering only happens when it's otherwise unoccupied."""
|
||||
self._advance_emote()
|
||||
self._wander_tick()
|
||||
|
||||
def _wander_tick(self) -> None:
|
||||
# Only stroll while genuinely idle: not mid-drag, not napping, not
|
||||
# talking/listening, and not while a speech bubble is up (it would walk
|
||||
# out from under it). A commanded `petctl move` ignores all of that
|
||||
# except the drag — if Bolt says go, it goes.
|
||||
busy = (
|
||||
not self._wander_enabled
|
||||
or self._napping
|
||||
or self._current_state != PetState.IDLE
|
||||
or self._bubble.isVisible()
|
||||
)
|
||||
if self._drag_offset is not None or (busy and not self._commanded_move):
|
||||
self._stop_walking()
|
||||
self._schedule_next_wander() # settle first, then wander
|
||||
return
|
||||
|
||||
if self._wander_target is None:
|
||||
if time.monotonic() < self._next_wander_at:
|
||||
return
|
||||
self._wander_target = self._pick_wander_target()
|
||||
if self._wander_target is None:
|
||||
self._schedule_next_wander()
|
||||
return
|
||||
|
||||
here = self.pos()
|
||||
dx = self._wander_target.x() - here.x()
|
||||
dy = self._wander_target.y() - here.y()
|
||||
distance = math.hypot(dx, dy)
|
||||
step = max(1.0, config.PET_WANDER_SPEED * _WANDER_TICK_MS / 1000.0)
|
||||
if distance <= step:
|
||||
self.move(self._wander_target)
|
||||
self._stop_walking()
|
||||
self._schedule_next_wander()
|
||||
self.snap_to_edge()
|
||||
else:
|
||||
self.move(round(here.x() + dx / distance * step), round(here.y() + dy / distance * step))
|
||||
self._bob_phase += 0.45 # little walk-cycle hop
|
||||
self._bob_offset = int(round(-2.5 * abs(math.sin(self._bob_phase))))
|
||||
self.update()
|
||||
self._reposition_bubble()
|
||||
|
||||
# ── state / speech ──────────────────────────────────────────────────
|
||||
|
||||
def set_state(self, state: PetState) -> None:
|
||||
if state == self._current_state:
|
||||
return
|
||||
self._current_state = state
|
||||
self.sprites.get(state).reset()
|
||||
if state != PetState.IDLE and not self._commanded_move:
|
||||
# Stand still while listening/thinking/talking — but not if Bolt
|
||||
# just told it to walk somewhere: that command arrives mid-turn,
|
||||
# and the reply (-> TALKING) lands a moment later.
|
||||
self._stop_walking()
|
||||
self.update()
|
||||
|
||||
def say(self, text: str, duration_ms: int = 6000) -> None:
|
||||
self._reposition_bubble()
|
||||
self._bubble.show_text(text, duration_ms)
|
||||
|
||||
def closeEvent(self, event) -> None:
|
||||
self._bubble.close()
|
||||
super().closeEvent(event)
|
||||
|
||||
# ── animation ────────────────────────────────────────────────────────
|
||||
|
||||
def _advance_frame(self) -> None:
|
||||
self.sprites.get(self._current_state).advance()
|
||||
self.update()
|
||||
|
||||
def paintEvent(self, _event) -> None:
|
||||
painter = QPainter(self)
|
||||
painter.setRenderHint(QPainter.Antialiasing)
|
||||
painter.setRenderHint(QPainter.SmoothPixmapTransform)
|
||||
pixmap: Optional[QPixmap] = self.sprites.get(self._current_state).current()
|
||||
if pixmap is None:
|
||||
self._apply_input_mask(None, 0, 0)
|
||||
return
|
||||
# Non-square source art (e.g. the Kenney robot sprites) keeps its
|
||||
# aspect ratio when scaled in SpriteSet, so it may be narrower or
|
||||
# shorter than the (square) window — center it either way.
|
||||
x = (self.width() - pixmap.width()) // 2
|
||||
y = (self.height() - pixmap.height()) // 2 + self._bob_offset
|
||||
# The input mask tracks the resting position, not the emote/bob
|
||||
# offset: rebuilding it every frame of a spin would be both expensive
|
||||
# and visibly janky, and the offsets are only a few pixels.
|
||||
self._apply_input_mask(pixmap, x, (self.height() - pixmap.height()) // 2)
|
||||
|
||||
dx, dy, angle, scale = self._emote_transform()
|
||||
if (dx, dy, angle, scale) == (0.0, 0.0, 0.0, 1.0):
|
||||
painter.drawPixmap(x, y, pixmap)
|
||||
return
|
||||
# Rotate/scale about the sprite's own center so a spin doesn't orbit
|
||||
# the window's corner.
|
||||
center_x = x + pixmap.width() / 2
|
||||
center_y = y + pixmap.height() / 2
|
||||
transform = QTransform()
|
||||
transform.translate(center_x + dx, center_y + dy)
|
||||
transform.rotate(angle)
|
||||
transform.scale(scale, scale)
|
||||
transform.translate(-pixmap.width() / 2, -pixmap.height() / 2)
|
||||
painter.setTransform(transform)
|
||||
painter.drawPixmap(0, 0, pixmap)
|
||||
|
||||
# ── drag / click-to-talk ─────────────────────────────────────────────
|
||||
|
||||
def mousePressEvent(self, event) -> None:
|
||||
if event.button() == Qt.LeftButton:
|
||||
global_pos = event.globalPosition().toPoint()
|
||||
self._drag_offset = global_pos - self.frameGeometry().topLeft()
|
||||
self._press_pos = global_pos
|
||||
self._dragged = False
|
||||
self._stop_walking() # grabbing it interrupts a stroll at once
|
||||
|
||||
def mouseMoveEvent(self, event) -> None:
|
||||
if self._drag_offset is None:
|
||||
return
|
||||
global_pos = event.globalPosition().toPoint()
|
||||
self.move(global_pos - self._drag_offset)
|
||||
self._reposition_bubble()
|
||||
if (global_pos - self._press_pos).manhattanLength() > _DRAG_THRESHOLD_PX:
|
||||
self._dragged = True
|
||||
|
||||
def mouseReleaseEvent(self, event) -> None:
|
||||
if event.button() != Qt.LeftButton:
|
||||
return
|
||||
was_click = not self._dragged
|
||||
self._drag_offset = None
|
||||
self._press_pos = None
|
||||
if was_click:
|
||||
self.talk_requested.emit()
|
||||
else:
|
||||
self.snap_to_edge() # dropped near an edge -> tuck it flush
|
||||
@@ -0,0 +1,111 @@
|
||||
"""Sprite loading + frame animation.
|
||||
|
||||
Convention: assets/sprites/<state>/*.png, frames played in filename-sorted
|
||||
order (e.g. frame_00.png, frame_01.png, ...), looping. <state> matches
|
||||
bolt_pet.state.PetState values: idle, listening, thinking, talking.
|
||||
|
||||
If a state's directory has no frames (real art not dropped in yet), falls
|
||||
back to a small procedurally-drawn placeholder blob so the app still runs
|
||||
end-to-end. Swap in real sprite sheets by pointing SPRITE_DIR at your own
|
||||
folder (see assets/sprites/README.md) — no code changes needed as long as
|
||||
the same per-state-subfolder-of-PNGs convention is followed. If your sheets
|
||||
use a different layout (single grid image, etc.), tell me the format and
|
||||
this loader can be adapted.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from PySide6.QtCore import QSize, Qt
|
||||
from PySide6.QtGui import QColor, QPainter, QPixmap
|
||||
|
||||
from ..state import PetState
|
||||
|
||||
DEFAULT_SPRITE_DIR = Path(__file__).resolve().parent.parent / "assets" / "sprites"
|
||||
|
||||
# Placeholder palette per state, used only when no frames are found.
|
||||
_PLACEHOLDER_COLORS = {
|
||||
PetState.IDLE: QColor(120, 170, 240),
|
||||
PetState.LISTENING: QColor(120, 220, 160),
|
||||
PetState.THINKING: QColor(230, 190, 90),
|
||||
PetState.TALKING: QColor(240, 130, 150),
|
||||
PetState.ERROR: QColor(220, 90, 90),
|
||||
}
|
||||
|
||||
|
||||
def _placeholder_frames(state: PetState, size: int) -> list[QPixmap]:
|
||||
"""A tiny 2-frame "breathing" blob so idle/listening/etc. are visually
|
||||
distinguishable even before real art exists."""
|
||||
color = _PLACEHOLDER_COLORS.get(state, QColor(150, 150, 150))
|
||||
frames = []
|
||||
for scale in (1.0, 0.92):
|
||||
pixmap = QPixmap(size, size)
|
||||
pixmap.fill(Qt.transparent)
|
||||
painter = QPainter(pixmap)
|
||||
painter.setRenderHint(QPainter.Antialiasing)
|
||||
painter.setBrush(color)
|
||||
painter.setPen(Qt.NoPen)
|
||||
margin = size * (1 - scale) / 2
|
||||
painter.drawEllipse(int(margin), int(margin), int(size * scale), int(size * scale))
|
||||
# simple eyes so it reads as a face, not just a circle
|
||||
eye_r = max(2, size // 16)
|
||||
eye_y = int(size * 0.42)
|
||||
painter.setBrush(QColor(30, 30, 40))
|
||||
painter.drawEllipse(int(size * 0.36) - eye_r, eye_y - eye_r, eye_r * 2, eye_r * 2)
|
||||
painter.drawEllipse(int(size * 0.64) - eye_r, eye_y - eye_r, eye_r * 2, eye_r * 2)
|
||||
painter.end()
|
||||
frames.append(pixmap)
|
||||
return frames
|
||||
|
||||
|
||||
class SpriteAnimation:
|
||||
"""One state's frame sequence + current playback position."""
|
||||
|
||||
def __init__(self, frames: list[QPixmap]):
|
||||
self.frames = frames or []
|
||||
self._index = 0
|
||||
|
||||
def advance(self) -> None:
|
||||
if self.frames:
|
||||
self._index = (self._index + 1) % len(self.frames)
|
||||
|
||||
def current(self) -> Optional[QPixmap]:
|
||||
if not self.frames:
|
||||
return None
|
||||
return self.frames[self._index]
|
||||
|
||||
def reset(self) -> None:
|
||||
self._index = 0
|
||||
|
||||
|
||||
def _load_frames_from_dir(directory: Path, size: int) -> list[QPixmap]:
|
||||
if not directory.is_dir():
|
||||
return []
|
||||
paths = sorted(directory.glob("*.png")) + sorted(directory.glob("*.PNG"))
|
||||
frames = []
|
||||
for path in paths:
|
||||
pixmap = QPixmap(str(path))
|
||||
if pixmap.isNull():
|
||||
continue
|
||||
if pixmap.size() != QSize(size, size):
|
||||
pixmap = pixmap.scaled(size, size, Qt.KeepAspectRatio, Qt.SmoothTransformation)
|
||||
frames.append(pixmap)
|
||||
return frames
|
||||
|
||||
|
||||
class SpriteSet:
|
||||
"""All animations for every PetState, loaded from *sprite_dir*."""
|
||||
|
||||
def __init__(self, sprite_dir: Path = DEFAULT_SPRITE_DIR, size: int = 160):
|
||||
self.size = size
|
||||
self._animations: dict[PetState, SpriteAnimation] = {}
|
||||
for state in PetState:
|
||||
frames = _load_frames_from_dir(sprite_dir / state.value, size)
|
||||
if not frames:
|
||||
frames = _placeholder_frames(state, size)
|
||||
self._animations[state] = SpriteAnimation(frames)
|
||||
|
||||
def get(self, state: PetState) -> SpriteAnimation:
|
||||
return self._animations[state]
|
||||
@@ -0,0 +1,133 @@
|
||||
"""System tray icon — the pet window is frameless with no taskbar entry, so
|
||||
this menu is the only always-available way to control or exit it: talk now,
|
||||
mute, wander, click-through, nap, history, wake-word tuning, quit.
|
||||
|
||||
Every entry is a plain callback passed in by ui/app.py; this file knows
|
||||
nothing about the controller or the pet window.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Callable, Optional
|
||||
|
||||
from PySide6.QtCore import Qt
|
||||
from PySide6.QtGui import QAction, QColor, QIcon, QPainter, QPixmap
|
||||
from PySide6.QtWidgets import QMenu, QSystemTrayIcon
|
||||
|
||||
|
||||
def _make_icon(muted: bool, napping: bool = False) -> QIcon:
|
||||
pixmap = QPixmap(32, 32)
|
||||
pixmap.fill(Qt.transparent)
|
||||
painter = QPainter(pixmap)
|
||||
painter.setRenderHint(QPainter.Antialiasing)
|
||||
if muted:
|
||||
color = QColor(200, 60, 60)
|
||||
elif napping:
|
||||
color = QColor(120, 120, 140)
|
||||
else:
|
||||
color = QColor(120, 170, 240)
|
||||
painter.setBrush(color)
|
||||
painter.setPen(Qt.NoPen)
|
||||
painter.drawEllipse(2, 2, 28, 28)
|
||||
painter.end()
|
||||
return QIcon(pixmap)
|
||||
|
||||
|
||||
class PetTray(QSystemTrayIcon):
|
||||
def __init__(
|
||||
self,
|
||||
on_talk_now: Callable[[], None],
|
||||
on_toggle_mute: Callable[[], bool],
|
||||
on_quit: Callable[[], None],
|
||||
on_set_wander: Optional[Callable[[bool], None]] = None,
|
||||
wander_enabled: bool = True,
|
||||
on_set_click_through: Optional[Callable[[bool], None]] = None,
|
||||
click_through_enabled: bool = False,
|
||||
on_set_nap: Optional[Callable[[bool], None]] = None,
|
||||
on_show_history: Optional[Callable[[], None]] = None,
|
||||
on_show_wake_tuner: Optional[Callable[[], None]] = None,
|
||||
parent=None,
|
||||
):
|
||||
super().__init__(_make_icon(muted=False), parent)
|
||||
self._on_toggle_mute = on_toggle_mute
|
||||
self._muted = False
|
||||
self._napping = False
|
||||
self.setToolTip("Bolt")
|
||||
|
||||
menu = QMenu()
|
||||
self._talk_action = QAction("Talk now", menu)
|
||||
self._talk_action.triggered.connect(on_talk_now)
|
||||
menu.addAction(self._talk_action)
|
||||
|
||||
self._mute_action = QAction("Mute mic", menu)
|
||||
self._mute_action.setCheckable(True)
|
||||
self._mute_action.triggered.connect(self._handle_toggle_mute)
|
||||
menu.addAction(self._mute_action)
|
||||
|
||||
self._nap_action = None
|
||||
if on_set_nap is not None:
|
||||
self._nap_action = QAction("Nap (no proactive noise)", menu)
|
||||
self._nap_action.setCheckable(True)
|
||||
self._nap_action.triggered.connect(lambda checked: on_set_nap(checked))
|
||||
menu.addAction(self._nap_action)
|
||||
|
||||
menu.addSeparator()
|
||||
|
||||
if on_set_wander is not None:
|
||||
self._wander_action = QAction("Wander around", menu)
|
||||
self._wander_action.setCheckable(True)
|
||||
self._wander_action.setChecked(wander_enabled)
|
||||
self._wander_action.triggered.connect(lambda checked: on_set_wander(checked))
|
||||
menu.addAction(self._wander_action)
|
||||
|
||||
if on_set_click_through is not None:
|
||||
self._click_through_action = QAction("Click through the pet", menu)
|
||||
self._click_through_action.setCheckable(True)
|
||||
self._click_through_action.setChecked(click_through_enabled)
|
||||
self._click_through_action.setToolTip(
|
||||
"Ignore the mouse entirely — control it from this menu instead."
|
||||
)
|
||||
self._click_through_action.triggered.connect(lambda checked: on_set_click_through(checked))
|
||||
menu.addAction(self._click_through_action)
|
||||
|
||||
menu.addSeparator()
|
||||
|
||||
if on_show_history is not None:
|
||||
history_action = QAction("History…", menu)
|
||||
history_action.triggered.connect(on_show_history)
|
||||
menu.addAction(history_action)
|
||||
|
||||
if on_show_wake_tuner is not None:
|
||||
tuner_action = QAction("Wake word tuning…", menu)
|
||||
tuner_action.triggered.connect(on_show_wake_tuner)
|
||||
menu.addAction(tuner_action)
|
||||
|
||||
menu.addSeparator()
|
||||
quit_action = QAction("Quit", menu)
|
||||
quit_action.triggered.connect(on_quit)
|
||||
menu.addAction(quit_action)
|
||||
|
||||
self.setContextMenu(menu)
|
||||
self.show()
|
||||
|
||||
def _handle_toggle_mute(self) -> None:
|
||||
self._muted = self._on_toggle_mute()
|
||||
self._mute_action.setChecked(self._muted)
|
||||
self._refresh_icon()
|
||||
|
||||
def set_napping(self, napping: bool) -> None:
|
||||
"""Reflect a nap the *controller* decided on (quiet hours, fullscreen,
|
||||
or a petctl command) — not just ones clicked here."""
|
||||
self._napping = napping
|
||||
if self._nap_action is not None:
|
||||
self._nap_action.setChecked(napping)
|
||||
self._refresh_icon()
|
||||
|
||||
def _refresh_icon(self) -> None:
|
||||
self.setIcon(_make_icon(self._muted, self._napping))
|
||||
if self._muted:
|
||||
self.setToolTip("Bolt (muted)")
|
||||
elif self._napping:
|
||||
self.setToolTip("Bolt (napping)")
|
||||
else:
|
||||
self.setToolTip("Bolt")
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Wake-word sensitivity tuner.
|
||||
|
||||
WAKE_WORD_THRESHOLD is otherwise a number you guess at in .env, restart, and
|
||||
then test by saying "thunderbolt" at your computer repeatedly. This window
|
||||
makes it evidence-based: a live peak-score readout while you talk, a rolling
|
||||
list of near misses (frames that scored just under the threshold — i.e. the
|
||||
times it *nearly* heard you), and a slider that takes effect immediately,
|
||||
mid-listen, without a restart.
|
||||
|
||||
The controller owns the threshold; this window is a view over it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from typing import Callable
|
||||
|
||||
from PySide6.QtCore import Qt, QTimer
|
||||
from PySide6.QtWidgets import (
|
||||
QDialog, QHBoxLayout, QLabel, QListWidget, QPushButton, QSlider, QVBoxLayout,
|
||||
)
|
||||
|
||||
_SLIDER_SCALE = 100 # QSlider is integer-only; threshold is 0.00-1.00
|
||||
|
||||
|
||||
class WakeTunerWindow(QDialog):
|
||||
def __init__(
|
||||
self,
|
||||
get_threshold: Callable[[], float],
|
||||
set_threshold: Callable[[float], None],
|
||||
get_stats: Callable[[], dict],
|
||||
on_reset: Callable[[], None],
|
||||
):
|
||||
super().__init__()
|
||||
self._get_threshold = get_threshold
|
||||
self._set_threshold = set_threshold
|
||||
self._get_stats = get_stats
|
||||
self._on_reset = on_reset
|
||||
|
||||
self.setWindowTitle("Bolt — wake word tuning")
|
||||
self.resize(460, 380)
|
||||
|
||||
self._threshold_label = QLabel()
|
||||
self._slider = QSlider(Qt.Horizontal)
|
||||
self._slider.setRange(5, 99)
|
||||
self._slider.setValue(int(round(get_threshold() * _SLIDER_SCALE)))
|
||||
self._slider.valueChanged.connect(self._threshold_changed)
|
||||
|
||||
self._peak_label = QLabel("Peak score since reset: —")
|
||||
self._peak_label.setToolTip(
|
||||
'Say "thunderbolt" a few times and watch this. Set the threshold '
|
||||
"just below the peak you can hit reliably."
|
||||
)
|
||||
|
||||
self._misses = QListWidget()
|
||||
|
||||
reset_button = QPushButton("Reset stats")
|
||||
reset_button.clicked.connect(self._reset)
|
||||
close_button = QPushButton("Close")
|
||||
close_button.clicked.connect(self.close)
|
||||
close_button.setDefault(True)
|
||||
|
||||
buttons = QHBoxLayout()
|
||||
buttons.addWidget(reset_button)
|
||||
buttons.addStretch(1)
|
||||
buttons.addWidget(close_button)
|
||||
|
||||
layout = QVBoxLayout(self)
|
||||
layout.addWidget(self._threshold_label)
|
||||
layout.addWidget(self._slider)
|
||||
layout.addWidget(self._peak_label)
|
||||
layout.addWidget(QLabel("Near misses (heard something, didn't quite fire):"))
|
||||
layout.addWidget(self._misses)
|
||||
layout.addLayout(buttons)
|
||||
|
||||
# Polled rather than signal-driven: scores arrive ~12x/second on the
|
||||
# audio thread, and a queued signal per frame to repaint a label is
|
||||
# more traffic than this is worth.
|
||||
self._timer = QTimer(self)
|
||||
self._timer.timeout.connect(self.refresh)
|
||||
self._update_threshold_label()
|
||||
|
||||
def _threshold_changed(self, value: int) -> None:
|
||||
self._set_threshold(value / _SLIDER_SCALE)
|
||||
self._update_threshold_label()
|
||||
|
||||
def _update_threshold_label(self) -> None:
|
||||
threshold = self._slider.value() / _SLIDER_SCALE
|
||||
self._threshold_label.setText(
|
||||
f"Threshold: {threshold:.2f} (lower = more sensitive, more false triggers)"
|
||||
)
|
||||
|
||||
def _reset(self) -> None:
|
||||
self._on_reset()
|
||||
self.refresh()
|
||||
|
||||
def refresh(self) -> None:
|
||||
stats = self._get_stats() or {}
|
||||
peak = stats.get("peak", 0.0)
|
||||
self._peak_label.setText(f"Peak score since reset: {peak:.3f}")
|
||||
self._misses.clear()
|
||||
for timestamp, score, threshold in reversed(stats.get("near_misses", [])):
|
||||
when = time.strftime("%H:%M:%S", time.localtime(timestamp))
|
||||
self._misses.addItem(f"{when} scored {score:.3f} (threshold {threshold:.2f})")
|
||||
if self._misses.count() == 0:
|
||||
self._misses.addItem("Nothing yet — say the wake phrase a few times.")
|
||||
|
||||
def show_refreshed(self) -> None:
|
||||
self._slider.setValue(int(round(self._get_threshold() * _SLIDER_SCALE)))
|
||||
self.refresh()
|
||||
self.show()
|
||||
self.raise_()
|
||||
self.activateWindow()
|
||||
self._timer.start(500)
|
||||
|
||||
def closeEvent(self, event) -> None:
|
||||
self._timer.stop()
|
||||
super().closeEvent(event)
|
||||
@@ -0,0 +1,37 @@
|
||||
# Desktop UI
|
||||
PySide6>=6.6
|
||||
|
||||
# Audio I/O (mic capture + speaker playback) — cross-platform via PortAudio.
|
||||
# On Linux you may also need the system package: sudo apt install libportaudio2
|
||||
sounddevice>=0.4.6
|
||||
numpy>=1.24
|
||||
|
||||
# HTTP client to the Bolt desk API
|
||||
requests>=2.31
|
||||
|
||||
# Wake-word detection (local, offline after first run) — runs the
|
||||
# custom-trained thunderbolt.onnx model shipped in this repo, same runtime
|
||||
# as the main repo's desk_client/bolt_desk.py (bolt.onnx). First use
|
||||
# downloads openwakeword's feature-extraction sub-models (~few MB, cached
|
||||
# under the package's own resources/ dir afterward) — needs internet once.
|
||||
openwakeword
|
||||
|
||||
# Offline TTS fallback if ElevenLabs isn't configured or a request fails.
|
||||
# Uses SAPI5 on Windows, NSSpeechSynthesizer on macOS, espeak on Linux
|
||||
# (Linux also needs: sudo apt install espeak-ng).
|
||||
pyttsx3>=2.90
|
||||
|
||||
# Global push-to-talk hotkey (PUSH_TO_TALK_HOTKEY). Optional: the pet
|
||||
# degrades to wake-word + tray + click if it's missing or if the session
|
||||
# won't allow a global key hook (most Wayland setups; macOS needs
|
||||
# Accessibility permission).
|
||||
pynput>=1.7
|
||||
|
||||
# Optional — only needed for scripts/slice_spritesheet.py (converting a
|
||||
# grid sprite sheet into the per-frame-PNG convention sprite.py expects).
|
||||
# Not imported by the app itself.
|
||||
Pillow>=10.0
|
||||
|
||||
# Test runner (tests/ — pure logic, no audio hardware or display needed;
|
||||
# run with QT_QPA_PLATFORM=offscreen).
|
||||
pytest>=8.0
|
||||
@@ -0,0 +1,11 @@
|
||||
@echo off
|
||||
REM Convenience launcher for Windows.
|
||||
cd /d "%~dp0"
|
||||
|
||||
if not exist .venv (
|
||||
python -m venv .venv
|
||||
.venv\Scripts\pip install --upgrade pip
|
||||
.venv\Scripts\pip install -r requirements.txt
|
||||
)
|
||||
|
||||
.venv\Scripts\python -m bolt_pet
|
||||
@@ -0,0 +1,12 @@
|
||||
#!/usr/bin/env bash
|
||||
# Convenience launcher for macOS/Linux.
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
if [ ! -d .venv ]; then
|
||||
python3 -m venv .venv
|
||||
./.venv/bin/pip install --upgrade pip
|
||||
./.venv/bin/pip install -r requirements.txt
|
||||
fi
|
||||
|
||||
exec ./.venv/bin/python -m bolt_pet
|
||||
@@ -0,0 +1,50 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Cut a grid-based sprite sheet (rows x cols of equal-size frames in one
|
||||
PNG) into the assets/sprites/<state>/frame_NN.png convention this project's
|
||||
sprite loader expects.
|
||||
|
||||
Usage:
|
||||
python scripts/slice_spritesheet.py idle_sheet.png assets/sprites/idle --cols 6 --rows 1
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def slice_sheet(sheet_path: Path, out_dir: Path, cols: int, rows: int) -> int:
|
||||
sheet = Image.open(sheet_path).convert("RGBA")
|
||||
frame_w = sheet.width // cols
|
||||
frame_h = sheet.height // rows
|
||||
if frame_w == 0 or frame_h == 0:
|
||||
raise ValueError(f"sheet is {sheet.width}x{sheet.height}, too small for {cols}x{rows} frames")
|
||||
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
count = 0
|
||||
for row in range(rows):
|
||||
for col in range(cols):
|
||||
box = (col * frame_w, row * frame_h, (col + 1) * frame_w, (row + 1) * frame_h)
|
||||
frame = sheet.crop(box)
|
||||
frame.save(out_dir / f"frame_{count:02d}.png")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("sheet", type=Path, help="path to the grid sprite sheet PNG")
|
||||
parser.add_argument("out_dir", type=Path, help="e.g. assets/sprites/idle")
|
||||
parser.add_argument("--cols", type=int, required=True)
|
||||
parser.add_argument("--rows", type=int, default=1)
|
||||
args = parser.parse_args()
|
||||
|
||||
count = slice_sheet(args.sheet, args.out_dir, args.cols, args.rows)
|
||||
print(f"Wrote {count} frames to {args.out_dir}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Barge-in detection, driven by a fake mic stream (no audio hardware)."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet.audio.barge_in import BargeInDetector
|
||||
|
||||
|
||||
class FakeStream:
|
||||
"""Yields frames of a given amplitude, mimicking sounddevice's
|
||||
(data, overflowed) 2-D int16 return shape."""
|
||||
|
||||
def __init__(self, amplitudes):
|
||||
self._amplitudes = list(amplitudes)
|
||||
|
||||
def read(self, frames):
|
||||
amplitude = self._amplitudes.pop(0) if self._amplitudes else 0
|
||||
data = np.full((frames, 1), amplitude, dtype=np.int16)
|
||||
return data, False
|
||||
|
||||
|
||||
def test_silence_never_interrupts():
|
||||
detector = BargeInDetector(FakeStream([0] * 20), threshold=1000, required_frames=3)
|
||||
assert not any(detector.check() for _ in range(20))
|
||||
|
||||
|
||||
def test_sustained_speech_interrupts_after_the_required_frames():
|
||||
detector = BargeInDetector(FakeStream([2000] * 5), threshold=1000, required_frames=3)
|
||||
assert detector.check() is False
|
||||
assert detector.check() is False
|
||||
assert detector.check() is True
|
||||
|
||||
|
||||
def test_a_single_thump_does_not_interrupt():
|
||||
# loud, quiet, loud, quiet ... never three in a row
|
||||
detector = BargeInDetector(FakeStream([2000, 0, 2000, 0, 2000, 0]), threshold=1000, required_frames=3)
|
||||
assert not any(detector.check() for _ in range(6))
|
||||
|
||||
|
||||
def test_counter_resets_after_a_quiet_frame():
|
||||
detector = BargeInDetector(FakeStream([2000, 2000, 0, 2000, 2000, 2000]), threshold=1000, required_frames=3)
|
||||
results = [detector.check() for _ in range(6)]
|
||||
assert results == [False, False, False, False, False, True]
|
||||
|
||||
|
||||
def test_reset_clears_progress():
|
||||
detector = BargeInDetector(FakeStream([2000] * 6), threshold=1000, required_frames=3)
|
||||
detector.check()
|
||||
detector.check()
|
||||
detector.reset()
|
||||
assert detector.check() is False
|
||||
assert detector.loud_frames == 1
|
||||
|
||||
|
||||
def test_a_mic_error_mid_playback_is_not_fatal():
|
||||
class BrokenStream:
|
||||
def read(self, frames):
|
||||
raise OSError("device disappeared")
|
||||
|
||||
detector = BargeInDetector(BrokenStream(), threshold=1000, required_frames=1)
|
||||
assert detector.check() is False
|
||||
@@ -0,0 +1,195 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
from bolt_pet import controller as controller_mod
|
||||
from bolt_pet.state import PetState
|
||||
|
||||
# A QApplication is required before any QObject with signals can be built.
|
||||
_app = QApplication.instance() or QApplication(["test"])
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def no_screen_probes(monkeypatch):
|
||||
# The real ones shell out to xprop/osascript — irrelevant here, and slow
|
||||
# (or hung) on a headless box.
|
||||
monkeypatch.setattr(controller_mod.screen_context, "context_for", lambda text: text)
|
||||
monkeypatch.setattr(controller_mod.screen_context, "is_fullscreen_active", lambda: False)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ctrl():
|
||||
return controller_mod.PetController()
|
||||
|
||||
|
||||
def _capture(signal):
|
||||
events = []
|
||||
signal.connect(lambda *a: events.append(a[0] if len(a) == 1 else a))
|
||||
return events
|
||||
|
||||
|
||||
# ── conversation turn ────────────────────────────────────────────────────────
|
||||
|
||||
def test_full_turn_happy_path(monkeypatch, ctrl):
|
||||
states = _capture(ctrl.state_changed)
|
||||
said = _capture(ctrl.said)
|
||||
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's the weather")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse", lambda text, on_command=None: "sunny and 72")
|
||||
spoken = []
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert states == ["listening", "thinking", "talking", "idle"]
|
||||
assert said == ["sunny and 72"]
|
||||
assert spoken == ["sunny and 72"]
|
||||
assert ctrl._state.state == PetState.IDLE
|
||||
|
||||
|
||||
def test_turn_with_nothing_heard_returns_to_idle_without_calling_server(monkeypatch, ctrl):
|
||||
states = _capture(ctrl.state_changed)
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: None)
|
||||
called = {"n": 0}
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: called.__setitem__("n", called["n"] + 1))
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert states == ["listening", "idle"]
|
||||
assert called["n"] == 0
|
||||
|
||||
|
||||
def test_turn_with_empty_transcript_returns_to_idle(monkeypatch, ctrl):
|
||||
states = _capture(ctrl.state_changed)
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "")
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert states == ["listening", "thinking", "idle"]
|
||||
|
||||
|
||||
def test_turn_with_stt_error_flashes_error_then_idle(monkeypatch, ctrl):
|
||||
states = _capture(ctrl.state_changed)
|
||||
logs = _capture(ctrl.log)
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
|
||||
def boom(pcm):
|
||||
raise controller_mod.stt.SttError("deepgram is down")
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", boom)
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert states == ["listening", "thinking", "error", "idle"]
|
||||
assert any("deepgram is down" in msg for msg in logs)
|
||||
|
||||
|
||||
def test_turn_with_server_error_flashes_error_then_idle(monkeypatch, ctrl):
|
||||
states = _capture(ctrl.state_changed)
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance", lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "hello")
|
||||
|
||||
def boom(text, on_command=None):
|
||||
raise controller_mod.server_client.ServerError("server is down")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse", boom)
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert states == ["listening", "thinking", "error", "idle"]
|
||||
|
||||
|
||||
# ── mute / talk-now / wake-or-click disambiguation ──────────────────────────
|
||||
|
||||
def test_toggle_mute_flips_and_returns_new_state(ctrl):
|
||||
assert ctrl.toggle_mute() is True
|
||||
assert ctrl.toggle_mute() is False
|
||||
|
||||
|
||||
def test_wait_for_wake_or_click_true_on_phrase_detection(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.wake_word, "listen_for_wake_word", lambda *a, **k: True)
|
||||
ctrl._stream = object()
|
||||
assert ctrl._wait_for_wake_or_click() is True
|
||||
|
||||
|
||||
def test_wait_for_wake_or_click_true_on_manual_trigger(monkeypatch, ctrl):
|
||||
# listen_for_wake_word returns False because should_continue() went
|
||||
# false (the talk_now flag got set) — controller must still recognize
|
||||
# this as "proceed", not "spurious wakeup".
|
||||
def fake_listen(stream, should_continue, on_tick=None, **kwargs):
|
||||
ctrl.request_talk_now()
|
||||
should_continue() # simulate the loop noticing the flag
|
||||
return False
|
||||
|
||||
monkeypatch.setattr(controller_mod.wake_word, "listen_for_wake_word", fake_listen)
|
||||
ctrl._stream = object()
|
||||
assert ctrl._wait_for_wake_or_click() is True
|
||||
assert not ctrl._talk_now.is_set() # cleared after being consumed
|
||||
|
||||
|
||||
def test_wait_for_wake_or_click_false_on_shutdown(monkeypatch, ctrl):
|
||||
def fake_listen(stream, should_continue, on_tick=None, **kwargs):
|
||||
ctrl.stop()
|
||||
return False
|
||||
monkeypatch.setattr(controller_mod.wake_word, "listen_for_wake_word", fake_listen)
|
||||
ctrl._stream = object()
|
||||
assert ctrl._wait_for_wake_or_click() is False
|
||||
|
||||
|
||||
# ── heartbeat / proactive announcements ─────────────────────────────────────
|
||||
|
||||
def test_heartbeat_speaks_a_pending_announcement_when_idle(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "HEARTBEAT_INTERVAL_SECONDS", 0)
|
||||
monkeypatch.setattr(controller_mod.server_client, "report_status", lambda: "don't forget your 3pm")
|
||||
spoken = []
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: (spoken.append(text), True)[1])
|
||||
said = _capture(ctrl.said)
|
||||
|
||||
ctrl._maybe_heartbeat()
|
||||
|
||||
assert spoken == ["don't forget your 3pm"]
|
||||
assert said == ["don't forget your 3pm"]
|
||||
assert ctrl._state.state == PetState.IDLE
|
||||
|
||||
|
||||
def test_heartbeat_does_nothing_when_not_idle(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "HEARTBEAT_INTERVAL_SECONDS", 0)
|
||||
ctrl._state.transition(PetState.LISTENING)
|
||||
called = {"n": 0}
|
||||
monkeypatch.setattr(controller_mod.server_client, "report_status",
|
||||
lambda: called.__setitem__("n", called["n"] + 1))
|
||||
|
||||
ctrl._maybe_heartbeat()
|
||||
|
||||
assert called["n"] == 0
|
||||
|
||||
|
||||
def test_heartbeat_respects_the_interval(monkeypatch, ctrl):
|
||||
# Deterministic fake clock — time.monotonic()'s absolute value is
|
||||
# arbitrary (often system uptime), so asserting behavior relative to it
|
||||
# without controlling it would be flaky.
|
||||
fake_now = {"t": 1000.0}
|
||||
monkeypatch.setattr(controller_mod.time, "monotonic", lambda: fake_now["t"])
|
||||
monkeypatch.setattr(controller_mod.config, "HEARTBEAT_INTERVAL_SECONDS", 60)
|
||||
called = {"n": 0}
|
||||
monkeypatch.setattr(controller_mod.server_client, "report_status",
|
||||
lambda: called.__setitem__("n", called["n"] + 1))
|
||||
|
||||
ctrl._maybe_heartbeat() # last_heartbeat starts at 0.0 -> elapsed is huge -> runs
|
||||
assert called["n"] == 1
|
||||
|
||||
fake_now["t"] += 10 # only 10s later — inside the 60s interval
|
||||
ctrl._maybe_heartbeat()
|
||||
assert called["n"] == 1 # skipped
|
||||
|
||||
fake_now["t"] += 60 # now well past the interval
|
||||
ctrl._maybe_heartbeat()
|
||||
assert called["n"] == 2
|
||||
@@ -0,0 +1,243 @@
|
||||
"""Controller-level wiring for the newer behaviours: petctl routing,
|
||||
barge-in follow-up, quiet hours, screen context, notification forwarding and
|
||||
the live wake threshold.
|
||||
|
||||
Needs a QApplication (signals), so run with QT_QPA_PLATFORM=offscreen.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
from bolt_pet import controller as controller_mod
|
||||
from bolt_pet.notifications import Notification
|
||||
from bolt_pet.state import PetState
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
_app = QApplication.instance() or QApplication(["test"])
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def no_screen_probes(monkeypatch):
|
||||
monkeypatch.setattr(controller_mod.screen_context, "is_fullscreen_active", lambda: False)
|
||||
monkeypatch.setattr(controller_mod.screen_context, "context_for", lambda text: text)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ctrl():
|
||||
return controller_mod.PetController()
|
||||
|
||||
|
||||
def _capture(signal):
|
||||
events = []
|
||||
signal.connect(lambda *a: events.append(a[0] if len(a) == 1 else a))
|
||||
return events
|
||||
|
||||
|
||||
# ── petctl routing ──────────────────────────────────────────────────────────
|
||||
|
||||
def test_petctl_commands_never_reach_the_shell(monkeypatch, ctrl):
|
||||
ran = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
|
||||
actions = _capture(ctrl.action)
|
||||
|
||||
output = ctrl._handle_command("petctl move top-left")
|
||||
|
||||
assert ran == []
|
||||
assert actions == [{"action": "move", "anchor": "top-left"}]
|
||||
assert "top-left" in output
|
||||
|
||||
|
||||
def test_ordinary_commands_still_run_locally(monkeypatch, ctrl):
|
||||
ran = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "run_local_command",
|
||||
lambda cmd: ran.append(cmd) or "[exit 0]\n")
|
||||
actions = _capture(ctrl.action)
|
||||
|
||||
ctrl._handle_command("df -h /")
|
||||
|
||||
assert ran == ["df -h /"]
|
||||
assert actions == []
|
||||
|
||||
|
||||
def test_bad_petctl_syntax_is_reported_back_not_executed(monkeypatch, ctrl):
|
||||
ran = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "run_local_command", lambda cmd: ran.append(cmd))
|
||||
output = ctrl._handle_command("petctl move sideways")
|
||||
assert ran == []
|
||||
assert "[pet]" in output
|
||||
|
||||
|
||||
def test_petctl_nap_also_flips_the_controller_state(ctrl):
|
||||
napping = _capture(ctrl.napping)
|
||||
ctrl._handle_command("petctl nap on")
|
||||
assert napping == [True]
|
||||
assert ctrl._napping is True
|
||||
|
||||
|
||||
# ── barge-in ────────────────────────────────────────────────────────────────
|
||||
|
||||
def test_interrupted_playback_queues_an_immediate_next_turn(monkeypatch, ctrl):
|
||||
logs = _capture(ctrl.log)
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: False) # interrupted
|
||||
|
||||
ctrl._speak("a very long explanation")
|
||||
|
||||
assert ctrl._talk_now.is_set() # loop picks the next turn up without a wake word
|
||||
assert any("Interrupted" in message for message in logs)
|
||||
|
||||
|
||||
def test_uninterrupted_playback_does_not_queue_a_turn(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
ctrl._speak("short answer")
|
||||
assert not ctrl._talk_now.is_set()
|
||||
|
||||
|
||||
def test_speech_is_recorded_in_the_history(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
ctrl._speak("**bold** reply")
|
||||
assert ctrl.history.last().text == "**bold** reply" # raw, for copy/paste
|
||||
|
||||
|
||||
# ── screen context ──────────────────────────────────────────────────────────
|
||||
|
||||
def test_the_active_window_rides_along_with_the_utterance(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance",
|
||||
lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "what's this error?")
|
||||
monkeypatch.setattr(controller_mod.screen_context, "context_for",
|
||||
lambda text: f"{text}\n\n[on screen right now: app.py]")
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: True)
|
||||
sent = []
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: sent.append(text) or "that's a KeyError")
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert "[on screen right now: app.py]" in sent[0]
|
||||
# ...but the *history* keeps what you actually said, not the annotation.
|
||||
assert ctrl.history.entries()[0].text == "what's this error?"
|
||||
|
||||
|
||||
# ── quiet hours / do-not-disturb ────────────────────────────────────────────
|
||||
|
||||
def test_quiet_hours_suppress_the_heartbeat(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "HEARTBEAT_INTERVAL_SECONDS", 0)
|
||||
monkeypatch.setattr(controller_mod.config, "QUIET_HOURS", "00:00-23:59")
|
||||
called = {"n": 0}
|
||||
monkeypatch.setattr(controller_mod.server_client, "report_status",
|
||||
lambda: called.__setitem__("n", called["n"] + 1))
|
||||
|
||||
ctrl._maybe_heartbeat()
|
||||
|
||||
assert called["n"] == 0
|
||||
assert ctrl._napping is True
|
||||
|
||||
|
||||
def test_fullscreen_triggers_do_not_disturb(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "QUIET_HOURS", "")
|
||||
monkeypatch.setattr(controller_mod.config, "DND_ON_FULLSCREEN", True)
|
||||
monkeypatch.setattr(controller_mod.screen_context, "is_fullscreen_active", lambda: True)
|
||||
napping = _capture(ctrl.napping)
|
||||
|
||||
ctrl._refresh_nap_state()
|
||||
|
||||
assert napping == [True]
|
||||
|
||||
|
||||
def test_a_manual_nap_overrides_the_schedule(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "QUIET_HOURS", "")
|
||||
ctrl.set_napping(True)
|
||||
ctrl._last_nap_check = 0.0
|
||||
ctrl._refresh_nap_state()
|
||||
assert ctrl._napping is True # the schedule doesn't wake it back up
|
||||
|
||||
ctrl.set_napping(None) # back on schedule
|
||||
ctrl._last_nap_check = 0.0
|
||||
ctrl._refresh_nap_state()
|
||||
assert ctrl._napping is False
|
||||
|
||||
|
||||
def test_napping_still_answers_when_spoken_to(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.mic, "record_utterance",
|
||||
lambda *a, **k: np.zeros(10, dtype=np.int16))
|
||||
monkeypatch.setattr(controller_mod.stt, "transcribe", lambda pcm: "you awake?")
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: "always")
|
||||
spoken = []
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
|
||||
ctrl.set_napping(True)
|
||||
|
||||
ctrl._handle_conversation_turn()
|
||||
|
||||
assert spoken == ["always"]
|
||||
|
||||
|
||||
# ── notification bridge ─────────────────────────────────────────────────────
|
||||
|
||||
def test_notifications_are_forwarded_and_spoken(monkeypatch, ctrl):
|
||||
monkeypatch.setattr(controller_mod.config, "NOTIFICATION_MIN_INTERVAL_SECONDS", 0)
|
||||
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
|
||||
sent, spoken = [], []
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: sent.append(text) or "your build is green")
|
||||
monkeypatch.setattr(controller_mod.tts, "speak",
|
||||
lambda text, on_error=None, should_stop=None: spoken.append(text) or True)
|
||||
|
||||
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
|
||||
ctrl._drain_notifications()
|
||||
|
||||
assert "CI: Build finished" in sent[0]
|
||||
assert spoken == ["your build is green"]
|
||||
assert ctrl._state.state == PetState.IDLE
|
||||
|
||||
|
||||
def test_filtered_out_notifications_are_never_queued(ctrl):
|
||||
ctrl._notification_gate = controller_mod.notifications.NotificationGate("deploy", 0)
|
||||
ctrl._queue_notification(Notification(app="Chat", summary="lunch?", body=""))
|
||||
assert ctrl._pending_notifications == []
|
||||
|
||||
|
||||
def test_notifications_are_not_forwarded_while_napping(monkeypatch, ctrl):
|
||||
ctrl._notification_gate = controller_mod.notifications.NotificationGate("", 0)
|
||||
called = {"n": 0}
|
||||
monkeypatch.setattr(controller_mod.server_client, "converse",
|
||||
lambda text, on_command=None: called.__setitem__("n", called["n"] + 1))
|
||||
ctrl.set_napping(True)
|
||||
|
||||
ctrl._queue_notification(Notification(app="CI", summary="Build finished", body=""))
|
||||
ctrl._drain_notifications()
|
||||
|
||||
assert called["n"] == 0
|
||||
|
||||
|
||||
# ── wake threshold ──────────────────────────────────────────────────────────
|
||||
|
||||
def test_threshold_is_live_and_clamped(ctrl):
|
||||
ctrl.set_wake_threshold(0.72)
|
||||
assert ctrl.wake_threshold() == pytest.approx(0.72)
|
||||
ctrl.set_wake_threshold(5)
|
||||
assert ctrl.wake_threshold() == pytest.approx(0.99)
|
||||
ctrl.set_wake_threshold(-1)
|
||||
assert ctrl.wake_threshold() == pytest.approx(0.01)
|
||||
|
||||
|
||||
def test_near_misses_are_recorded_for_the_tuner(ctrl):
|
||||
ctrl.set_wake_threshold(0.5)
|
||||
ctrl._observe_wake_score(0.42, 0.5) # near miss
|
||||
ctrl._observe_wake_score(0.01, 0.5) # background noise, not interesting
|
||||
stats = ctrl.wake_stats()
|
||||
assert len(stats["near_misses"]) == 1
|
||||
assert stats["peak"] == pytest.approx(0.42)
|
||||
|
||||
ctrl.reset_wake_stats()
|
||||
assert ctrl.wake_stats()["near_misses"] == []
|
||||
@@ -0,0 +1,92 @@
|
||||
"""Conversation scrollback + push-to-talk hotkey parsing."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import history as history_mod
|
||||
from bolt_pet.hotkey import GlobalHotkey, HotkeyError, to_pynput_spec
|
||||
|
||||
|
||||
# ── history ─────────────────────────────────────────────────────────────────
|
||||
|
||||
def test_keeps_entries_in_order():
|
||||
log = history_mod.ConversationHistory(limit=10)
|
||||
log.add(history_mod.USER, "what's the weather")
|
||||
log.add(history_mod.PET, "sunny and 72")
|
||||
assert [e.text for e in log.entries()] == ["what's the weather", "sunny and 72"]
|
||||
|
||||
|
||||
def test_drops_the_oldest_past_the_limit():
|
||||
log = history_mod.ConversationHistory(limit=2)
|
||||
for i in range(5):
|
||||
log.add(history_mod.PET, f"line {i}")
|
||||
assert [e.text for e in log.entries()] == ["line 3", "line 4"]
|
||||
|
||||
|
||||
def test_blank_entries_are_ignored():
|
||||
log = history_mod.ConversationHistory()
|
||||
assert log.add(history_mod.PET, " ") is None
|
||||
assert len(log) == 0
|
||||
|
||||
|
||||
def test_last_can_filter_by_role():
|
||||
log = history_mod.ConversationHistory()
|
||||
log.add(history_mod.USER, "hello")
|
||||
log.add(history_mod.PET, "hi there")
|
||||
log.add(history_mod.USER, "still there?")
|
||||
assert log.last().text == "still there?"
|
||||
assert log.last(history_mod.PET).text == "hi there"
|
||||
|
||||
|
||||
def test_as_text_is_copyable_transcript():
|
||||
log = history_mod.ConversationHistory()
|
||||
log.add(history_mod.USER, "ping")
|
||||
log.add(history_mod.PET, "pong")
|
||||
assert log.as_text() == "You: ping\nBolt: pong"
|
||||
|
||||
|
||||
def test_timestamps_are_rendered_when_present():
|
||||
log = history_mod.ConversationHistory()
|
||||
log.add(history_mod.PET, "pong", timestamp=1710000000.0)
|
||||
assert log.as_text(clock=lambda t: "12:00:00") == "[12:00:00] Bolt: pong"
|
||||
|
||||
|
||||
def test_clear_empties_the_log():
|
||||
log = history_mod.ConversationHistory()
|
||||
log.add(history_mod.PET, "pong")
|
||||
log.clear()
|
||||
assert len(log) == 0
|
||||
|
||||
|
||||
# ── hotkey ──────────────────────────────────────────────────────────────────
|
||||
|
||||
def test_translates_a_readable_spec_to_pynput_syntax():
|
||||
assert to_pynput_spec("ctrl+alt+space") == "<ctrl>+<alt>+<space>"
|
||||
assert to_pynput_spec("ctrl+shift+b") == "<ctrl>+<shift>+b"
|
||||
|
||||
|
||||
def test_accepts_the_names_people_actually_type():
|
||||
assert to_pynput_spec("Control+Option+Space") == "<ctrl>+<alt>+<space>"
|
||||
assert to_pynput_spec("super+k") == "<cmd>+k"
|
||||
|
||||
|
||||
def test_empty_spec_is_an_error_at_parse_time():
|
||||
with pytest.raises(HotkeyError):
|
||||
to_pynput_spec("")
|
||||
|
||||
|
||||
def test_disabled_hotkey_starts_cleanly_and_reports_nothing():
|
||||
hotkey = GlobalHotkey("", lambda: None)
|
||||
assert hotkey.start() is None
|
||||
assert hotkey.running is False
|
||||
|
||||
|
||||
def test_invalid_hotkey_reports_instead_of_raising():
|
||||
hotkey = GlobalHotkey("+++", lambda: None)
|
||||
problem = hotkey.start()
|
||||
assert problem and "invalid" in problem
|
||||
assert hotkey.running is False
|
||||
@@ -0,0 +1,114 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from bolt_pet.audio import mic
|
||||
|
||||
FRAME_LEN = 320 # small for fast tests
|
||||
SAMPLE_RATE = 8000
|
||||
|
||||
|
||||
class _ScriptedStream:
|
||||
"""Replays a fixed list of frames, then quiet forever."""
|
||||
|
||||
def __init__(self, frames):
|
||||
self._frames = list(frames)
|
||||
|
||||
def read(self, frames):
|
||||
if self._frames:
|
||||
frame = self._frames.pop(0)
|
||||
else:
|
||||
frame = np.zeros(FRAME_LEN, dtype=np.int16)
|
||||
return frame.reshape(-1, 1), False
|
||||
|
||||
|
||||
def _loud(n=1):
|
||||
return [np.full(FRAME_LEN, 5000, dtype=np.int16) for _ in range(n)]
|
||||
|
||||
|
||||
def _quiet(n=1):
|
||||
return [np.zeros(FRAME_LEN, dtype=np.int16) for _ in range(n)]
|
||||
|
||||
|
||||
def test_returns_none_when_nothing_ever_gets_loud():
|
||||
stream = _ScriptedStream(_quiet(50))
|
||||
calls = {"i": 0}
|
||||
|
||||
def should_continue():
|
||||
calls["i"] += 1
|
||||
return calls["i"] <= 50
|
||||
|
||||
result = mic.record_utterance(
|
||||
stream, should_continue=should_continue,
|
||||
rms_threshold=300, silence_end_sec=0.5, max_utterance_s=5, min_utterance_s=0.1,
|
||||
frame_len=FRAME_LEN, sample_rate=SAMPLE_RATE,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
def test_captures_speech_and_stops_after_trailing_silence():
|
||||
# speech, then enough silence to cross the silence_end_sec threshold
|
||||
silence_end_sec = 0.5
|
||||
silence_limit_frames = int(silence_end_sec * SAMPLE_RATE / FRAME_LEN)
|
||||
frames = _loud(5) + _quiet(silence_limit_frames + 2)
|
||||
stream = _ScriptedStream(frames)
|
||||
|
||||
result = mic.record_utterance(
|
||||
stream, rms_threshold=300, silence_end_sec=silence_end_sec,
|
||||
max_utterance_s=5, min_utterance_s=0.05,
|
||||
frame_len=FRAME_LEN, sample_rate=SAMPLE_RATE,
|
||||
)
|
||||
assert result is not None
|
||||
# captured the loud frames plus the silence up to (and including) the
|
||||
# frame that crossed the silence-end threshold, but not endless silence
|
||||
assert len(result) < len(frames) * FRAME_LEN
|
||||
|
||||
|
||||
def test_returns_none_if_utterance_shorter_than_minimum():
|
||||
frames = _loud(1) + _quiet(2) # crosses silence limit almost immediately
|
||||
stream = _ScriptedStream(frames)
|
||||
result = mic.record_utterance(
|
||||
stream, rms_threshold=300, silence_end_sec=0.05,
|
||||
max_utterance_s=5, min_utterance_s=5.0, # impossible to satisfy
|
||||
frame_len=FRAME_LEN, sample_rate=SAMPLE_RATE,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
def test_stops_at_max_utterance_even_without_silence():
|
||||
max_utterance_s = 0.5
|
||||
max_frames = int(max_utterance_s * SAMPLE_RATE / FRAME_LEN)
|
||||
stream = _ScriptedStream(_loud(max_frames + 20)) # never goes quiet
|
||||
|
||||
result = mic.record_utterance(
|
||||
stream, rms_threshold=300, silence_end_sec=10.0, # would never trigger
|
||||
max_utterance_s=max_utterance_s, min_utterance_s=0.01,
|
||||
frame_len=FRAME_LEN, sample_rate=SAMPLE_RATE,
|
||||
)
|
||||
assert result is not None
|
||||
assert len(result) == max_frames * FRAME_LEN
|
||||
|
||||
|
||||
def test_returns_none_when_should_continue_stops_before_speech():
|
||||
stream = _ScriptedStream(_quiet(100))
|
||||
result = mic.record_utterance(stream, should_continue=lambda: False,
|
||||
frame_len=FRAME_LEN, sample_rate=SAMPLE_RATE)
|
||||
assert result is None
|
||||
|
||||
|
||||
def test_pcm_to_wav_bytes_round_trips_via_wave_module():
|
||||
import wave
|
||||
import io
|
||||
|
||||
pcm = np.array([0, 100, -100, 32767, -32768], dtype=np.int16)
|
||||
wav_bytes = mic.pcm_to_wav_bytes(pcm, sample_rate=16000)
|
||||
with wave.open(io.BytesIO(wav_bytes), "rb") as wf:
|
||||
assert wf.getnchannels() == 1
|
||||
assert wf.getsampwidth() == 2
|
||||
assert wf.getframerate() == 16000
|
||||
frames = wf.readframes(wf.getnframes())
|
||||
assert np.frombuffer(frames, dtype=np.int16).tolist() == pcm.tolist()
|
||||
@@ -0,0 +1,90 @@
|
||||
"""dbus-monitor parsing + the forward/rate-limit gate. No session bus
|
||||
needed — the parser is fed canned dbus-monitor output."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet.notifications import Notification, NotificationGate, iter_notifications
|
||||
|
||||
SAMPLE = '''signal time=1710000000.1 sender=org.freedesktop.DBus -> destination=:1.7 serial=2 path=/org/freedesktop/DBus; interface=org.freedesktop.DBus; member=NameAcquired
|
||||
string ":1.7"
|
||||
method call time=1710000001.2 sender=:1.72 -> destination=org.freedesktop.Notifications serial=88 path=/org/freedesktop/Notifications; interface=org.freedesktop.Notifications; member=Notify
|
||||
string "Firefox"
|
||||
uint32 0
|
||||
string ""
|
||||
string "Build finished"
|
||||
string "All 42 tests passed"
|
||||
array [
|
||||
]
|
||||
int32 -1
|
||||
method call time=1710000002.3 sender=:1.80 -> destination=org.freedesktop.Notifications serial=91 path=/org/freedesktop/Notifications; interface=org.freedesktop.Notifications; member=Notify
|
||||
string "Calendar"
|
||||
uint32 0
|
||||
string "calendar-icon"
|
||||
string "Standup in 5 minutes"
|
||||
string ""
|
||||
array [
|
||||
]
|
||||
'''
|
||||
|
||||
|
||||
def _parse(text=SAMPLE):
|
||||
return list(iter_notifications(text.splitlines()))
|
||||
|
||||
|
||||
def test_parses_each_notify_call():
|
||||
parsed = _parse()
|
||||
assert parsed == [
|
||||
Notification(app="Firefox", summary="Build finished", body="All 42 tests passed"),
|
||||
Notification(app="Calendar", summary="Standup in 5 minutes", body=""),
|
||||
]
|
||||
|
||||
|
||||
def test_unrelated_dbus_traffic_is_ignored():
|
||||
noise = SAMPLE.split("method call")[0]
|
||||
assert _parse(noise) == []
|
||||
|
||||
|
||||
def test_trailing_notification_without_a_following_block_is_still_emitted():
|
||||
assert _parse()[-1].summary == "Standup in 5 minutes"
|
||||
|
||||
|
||||
def test_as_text_is_what_gets_sent_to_the_server():
|
||||
assert _parse()[0].as_text() == "Firefox: Build finished — All 42 tests passed"
|
||||
assert _parse()[1].as_text() == "Calendar: Standup in 5 minutes"
|
||||
|
||||
|
||||
# ── gate ────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _notification(summary="Build finished", app="Firefox"):
|
||||
return Notification(app=app, summary=summary, body="")
|
||||
|
||||
|
||||
def test_empty_filter_forwards_everything():
|
||||
gate = NotificationGate("", min_interval=0)
|
||||
assert gate.should_forward(_notification(), now=0) is True
|
||||
|
||||
|
||||
def test_filter_regex_selects_what_is_worth_a_round_trip():
|
||||
gate = NotificationGate(r"build|deploy", min_interval=0)
|
||||
assert gate.should_forward(_notification("Build finished"), now=0) is True
|
||||
assert gate.should_forward(_notification("New message from Dave"), now=1) is False
|
||||
|
||||
|
||||
def test_rate_limit_drops_a_burst():
|
||||
gate = NotificationGate("", min_interval=60)
|
||||
assert gate.should_forward(_notification(), now=100) is True
|
||||
assert gate.should_forward(_notification(), now=120) is False
|
||||
assert gate.should_forward(_notification(), now=161) is True
|
||||
|
||||
|
||||
def test_a_broken_regex_does_not_silence_the_bridge():
|
||||
gate = NotificationGate("(unclosed", min_interval=0)
|
||||
assert gate.should_forward(_notification(), now=0) is True
|
||||
|
||||
|
||||
def test_empty_notifications_are_dropped():
|
||||
gate = NotificationGate("", min_interval=0)
|
||||
assert gate.should_forward(Notification(app="", summary="", body=""), now=0) is False
|
||||
@@ -0,0 +1,71 @@
|
||||
"""petctl parsing — the pseudo-commands the server can relay to drive the
|
||||
pet's body instead of a shell."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import pet_actions
|
||||
|
||||
|
||||
def test_non_pet_commands_are_left_alone():
|
||||
assert pet_actions.parse("ls -la") is None
|
||||
assert pet_actions.parse("systemctl restart nginx") is None
|
||||
assert pet_actions.parse("") is None
|
||||
# "petstore" must not be mistaken for the "pet" prefix
|
||||
assert pet_actions.parse("petstore --list") is None
|
||||
|
||||
|
||||
def test_move_to_an_anchor():
|
||||
assert pet_actions.parse("petctl move top-left") == {"action": "move", "anchor": "top-left"}
|
||||
assert pet_actions.parse("petctl move bottom_right") == {"action": "move", "anchor": "bottom-right"}
|
||||
|
||||
|
||||
def test_move_to_coordinates():
|
||||
assert pet_actions.parse("petctl move 300 120") == {"action": "move", "x": 300, "y": 120}
|
||||
|
||||
|
||||
def test_move_rejects_nonsense_targets():
|
||||
with pytest.raises(pet_actions.ActionError):
|
||||
pet_actions.parse("petctl move sideways")
|
||||
with pytest.raises(pet_actions.ActionError):
|
||||
pet_actions.parse("petctl move")
|
||||
|
||||
|
||||
def test_emotes():
|
||||
assert pet_actions.parse("petctl emote wave") == {"action": "emote", "emote": "wave"}
|
||||
with pytest.raises(pet_actions.ActionError):
|
||||
pet_actions.parse("petctl emote moonwalk")
|
||||
|
||||
|
||||
def test_say_keeps_the_whole_sentence():
|
||||
assert pet_actions.parse('petctl say "build is green"') == {
|
||||
"action": "say", "text": "build is green"
|
||||
}
|
||||
assert pet_actions.parse("petctl say build is green")["text"] == "build is green"
|
||||
|
||||
|
||||
def test_wander_and_nap_toggles():
|
||||
assert pet_actions.parse("petctl wander off") == {"action": "wander", "enabled": False}
|
||||
assert pet_actions.parse("petctl nap on") == {"action": "nap", "enabled": True}
|
||||
with pytest.raises(pet_actions.ActionError):
|
||||
pet_actions.parse("petctl wander maybe")
|
||||
|
||||
|
||||
def test_alternate_prefixes_and_verbs():
|
||||
assert pet_actions.parse("bolt-pet goto center")["anchor"] == "center"
|
||||
assert pet_actions.parse("pet do hop")["emote"] == "hop"
|
||||
|
||||
|
||||
def test_unknown_verb_is_an_error_not_a_shell_command():
|
||||
with pytest.raises(pet_actions.ActionError):
|
||||
pet_actions.parse("petctl explode")
|
||||
|
||||
|
||||
def test_describe_is_reported_back_to_the_server():
|
||||
assert "top-left" in pet_actions.describe({"action": "move", "anchor": "top-left"})
|
||||
assert "wave" in pet_actions.describe({"action": "emote", "emote": "wave"})
|
||||
assert pet_actions.describe({"action": "help"}) == pet_actions.HELP
|
||||
@@ -0,0 +1,181 @@
|
||||
"""Window-side behaviour for the newer features: emote curves, petctl
|
||||
actions, edge snapping, napping, click-through.
|
||||
|
||||
Needs a QApplication — run with QT_QPA_PLATFORM=offscreen.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from PySide6.QtCore import QPoint
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import config
|
||||
from bolt_pet.state import PetState
|
||||
from bolt_pet.ui.pet_window import _EMOTE_TICKS, PetWindow, emote_transform
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def qt_app():
|
||||
yield QApplication.instance() or QApplication([])
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pet(qt_app):
|
||||
window = PetWindow()
|
||||
yield window
|
||||
window.close()
|
||||
|
||||
|
||||
# ── emote curves (pure maths) ───────────────────────────────────────────────
|
||||
|
||||
@pytest.mark.parametrize("emote", ["wave", "hop", "bounce", "spin", "nod", "shake", "wiggle"])
|
||||
def test_every_emote_returns_the_sprite_to_rest(emote):
|
||||
# Anything that doesn't land back at the identity transform leaves the pet
|
||||
# permanently askew. (approx: the sine curves land on ~1e-16, not 0.0.)
|
||||
rest = pytest.approx((0.0, 0.0, 0.0, 1.0), abs=1e-9)
|
||||
assert emote_transform(emote, 1.0) == rest
|
||||
assert emote_transform(emote, 0.0) == rest
|
||||
|
||||
|
||||
def test_emotes_actually_move_the_sprite_mid_animation():
|
||||
for emote in ("wave", "hop", "spin", "nod", "shake"):
|
||||
samples = [emote_transform(emote, i / 20) for i in range(1, 20)]
|
||||
assert any(sample != (0.0, 0.0, 0.0, 1.0) for sample in samples), emote
|
||||
|
||||
|
||||
def test_an_unknown_emote_is_a_no_op_not_a_crash():
|
||||
assert emote_transform("moonwalk", 0.5) == (0.0, 0.0, 0.0, 1.0)
|
||||
|
||||
|
||||
def test_progress_is_clamped():
|
||||
assert emote_transform("hop", 5.0) == emote_transform("hop", 1.0)
|
||||
assert emote_transform("hop", -3.0) == emote_transform("hop", 0.0)
|
||||
|
||||
|
||||
def test_an_emote_finishes_and_clears_itself(pet):
|
||||
pet.start_emote("spin")
|
||||
assert pet._emote == "spin"
|
||||
for _ in range(_EMOTE_TICKS + 2):
|
||||
pet._advance_emote()
|
||||
assert pet._emote is None
|
||||
|
||||
|
||||
# ── petctl actions ──────────────────────────────────────────────────────────
|
||||
|
||||
def test_move_action_sets_a_walk_target(pet):
|
||||
pet.apply_action({"action": "move", "anchor": "top-left"})
|
||||
assert pet._wander_target is not None
|
||||
assert pet._commanded_move is True
|
||||
|
||||
|
||||
def test_commanded_moves_happen_even_while_talking(pet, monkeypatch):
|
||||
monkeypatch.setattr(config, "PET_EDGE_SNAP", False) # snapping would move it again on arrival
|
||||
pet.set_state(PetState.TALKING)
|
||||
pet.apply_action({"action": "move", "anchor": "top-left"})
|
||||
target = pet._wander_target
|
||||
if target is None:
|
||||
pytest.skip("no usable screen geometry on this host")
|
||||
for _ in range(2000):
|
||||
pet._wander_tick()
|
||||
if pet._wander_target is None:
|
||||
break
|
||||
assert pet.pos() == target
|
||||
|
||||
|
||||
def test_a_commanded_move_survives_the_reply_arriving(pet):
|
||||
# Real ordering: `petctl move` comes back as a tool call mid-turn, then
|
||||
# the reply flips the pet to TALKING a moment later. That must not cancel
|
||||
# the walk it was just told to make.
|
||||
pet.apply_action({"action": "move", "anchor": "center"})
|
||||
pet.set_state(PetState.TALKING)
|
||||
assert pet._wander_target is not None
|
||||
|
||||
|
||||
def test_move_to_explicit_coordinates_is_clamped_on_screen(pet):
|
||||
pet.apply_action({"action": "move", "x": -5000, "y": -5000})
|
||||
geo = pet._screen_geometry()
|
||||
if geo is None:
|
||||
pytest.skip("no usable screen geometry on this host")
|
||||
assert pet._wander_target.x() >= geo.left()
|
||||
assert pet._wander_target.y() >= geo.top()
|
||||
|
||||
|
||||
def test_say_action_shows_the_bubble(pet):
|
||||
pet.apply_action({"action": "say", "text": "build is green"})
|
||||
assert pet._bubble.text == "build is green"
|
||||
|
||||
|
||||
def test_wander_and_nap_actions(pet):
|
||||
pet.apply_action({"action": "wander", "enabled": False})
|
||||
assert pet._wander_enabled is False
|
||||
pet.apply_action({"action": "nap", "enabled": True})
|
||||
assert pet.napping is True
|
||||
|
||||
|
||||
def test_emote_action_starts_the_emote(pet):
|
||||
pet.apply_action({"action": "emote", "emote": "wave"})
|
||||
assert pet._emote == "wave"
|
||||
|
||||
|
||||
# ── napping ─────────────────────────────────────────────────────────────────
|
||||
|
||||
def test_napping_dims_the_pet_and_stops_it_wandering(pet):
|
||||
pet.set_napping(True)
|
||||
assert pet.windowOpacity() < 1.0
|
||||
start = pet.pos()
|
||||
pet.wander_now()
|
||||
for _ in range(60):
|
||||
pet._wander_tick()
|
||||
assert pet.pos() == start
|
||||
|
||||
pet.set_napping(False)
|
||||
assert pet.windowOpacity() == 1.0
|
||||
|
||||
|
||||
# ── edge snapping ───────────────────────────────────────────────────────────
|
||||
|
||||
def test_snaps_flush_when_parked_near_an_edge(pet, monkeypatch):
|
||||
geo = pet._screen_geometry()
|
||||
if geo is None:
|
||||
pytest.skip("no usable screen geometry on this host")
|
||||
monkeypatch.setattr(config, "PET_EDGE_SNAP", True)
|
||||
pet.move(geo.left() + 10, geo.top() + 10)
|
||||
assert pet.snap_to_edge() is True
|
||||
assert pet.pos() == QPoint(geo.left(), geo.top())
|
||||
|
||||
|
||||
def test_does_not_snap_from_the_middle_of_the_screen(pet, monkeypatch):
|
||||
geo = pet._screen_geometry()
|
||||
if geo is None:
|
||||
pytest.skip("no usable screen geometry on this host")
|
||||
monkeypatch.setattr(config, "PET_EDGE_SNAP", True)
|
||||
middle = QPoint(geo.left() + geo.width() // 2, geo.top() + geo.height() // 2)
|
||||
pet.move(middle)
|
||||
assert pet.snap_to_edge() is False
|
||||
assert pet.pos() == middle
|
||||
|
||||
|
||||
def test_snapping_can_be_turned_off(pet, monkeypatch):
|
||||
geo = pet._screen_geometry()
|
||||
if geo is None:
|
||||
pytest.skip("no usable screen geometry on this host")
|
||||
monkeypatch.setattr(config, "PET_EDGE_SNAP", False)
|
||||
pet.move(geo.left() + 10, geo.top() + 10)
|
||||
assert pet.snap_to_edge() is False
|
||||
|
||||
|
||||
# ── click-through ───────────────────────────────────────────────────────────
|
||||
|
||||
def test_click_through_toggles_mouse_transparency(pet):
|
||||
from PySide6.QtCore import Qt
|
||||
|
||||
pet.set_click_through(True)
|
||||
assert pet.click_through is True
|
||||
assert pet.testAttribute(Qt.WA_TransparentForMouseEvents) is True
|
||||
|
||||
pet.set_click_through(False)
|
||||
assert pet.testAttribute(Qt.WA_TransparentForMouseEvents) is False
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Quiet-hours parsing and matching, without waiting for 11pm."""
|
||||
|
||||
import sys
|
||||
from datetime import time as dtime
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import quiet
|
||||
|
||||
|
||||
def test_no_spec_means_never_quiet():
|
||||
assert quiet.is_quiet("") is False
|
||||
assert quiet.is_quiet(" ") is False
|
||||
|
||||
|
||||
def test_simple_daytime_range():
|
||||
ranges = quiet.parse_ranges("13:00-14:00")
|
||||
assert quiet.in_ranges(dtime(13, 30), ranges) is True
|
||||
assert quiet.in_ranges(dtime(12, 59), ranges) is False
|
||||
assert quiet.in_ranges(dtime(14, 0), ranges) is False # end is exclusive
|
||||
|
||||
|
||||
def test_range_wrapping_past_midnight():
|
||||
ranges = quiet.parse_ranges("23:00-08:00")
|
||||
for moment in (dtime(23, 0), dtime(23, 59), dtime(0, 0), dtime(7, 59)):
|
||||
assert quiet.in_ranges(moment, ranges) is True, moment
|
||||
for moment in (dtime(8, 0), dtime(12, 0), dtime(22, 59)):
|
||||
assert quiet.in_ranges(moment, ranges) is False, moment
|
||||
|
||||
|
||||
def test_multiple_ranges():
|
||||
ranges = quiet.parse_ranges("23:00-08:00, 13:00-14:00")
|
||||
assert quiet.in_ranges(dtime(13, 15), ranges) is True
|
||||
assert quiet.in_ranges(dtime(2, 0), ranges) is True
|
||||
assert quiet.in_ranges(dtime(16, 0), ranges) is False
|
||||
|
||||
|
||||
def test_zero_length_range_is_not_all_day():
|
||||
assert quiet.in_ranges(dtime(12, 0), quiet.parse_ranges("09:00-09:00")) is False
|
||||
|
||||
|
||||
def test_malformed_specs_raise_when_parsed_directly():
|
||||
for spec in ("nonsense", "25:00-26:00", "13:00", "13:60-14:00"):
|
||||
with pytest.raises(quiet.QuietHoursError):
|
||||
quiet.parse_ranges(spec)
|
||||
|
||||
|
||||
def test_malformed_spec_is_reported_but_never_mutes_the_pet():
|
||||
seen = []
|
||||
assert quiet.is_quiet("nonsense", now=dtime(3, 0), on_error=seen.append) is False
|
||||
assert len(seen) == 1
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Screen-context parsing/annotation. The subprocess probes are platform
|
||||
specific; the parsing they feed is not, so that's what's tested here."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet import screen_context
|
||||
|
||||
|
||||
def test_parses_the_active_window_id():
|
||||
output = "_NET_ACTIVE_WINDOW(WINDOW): window id # 0x3c00007\n"
|
||||
assert screen_context.parse_xprop_window_id(output) == "0x3c00007"
|
||||
|
||||
|
||||
def test_no_active_window_id_when_nothing_is_focused():
|
||||
assert screen_context.parse_xprop_window_id("_NET_ACTIVE_WINDOW(WINDOW): window id # 0x0") is None
|
||||
assert screen_context.parse_xprop_window_id("") is None
|
||||
|
||||
|
||||
def test_parses_the_window_title():
|
||||
output = '_NET_WM_NAME(UTF8_STRING) = "bolt_pet/controller.py - Cursor"\n'
|
||||
assert screen_context.parse_xprop_window_name(output) == "bolt_pet/controller.py - Cursor"
|
||||
|
||||
|
||||
def test_unset_title_property():
|
||||
assert screen_context.parse_xprop_window_name("_NET_WM_NAME: not found") is None
|
||||
|
||||
|
||||
def test_fullscreen_state_detection():
|
||||
assert screen_context.parse_xprop_fullscreen(
|
||||
"_NET_WM_STATE(ATOM) = _NET_WM_STATE_FULLSCREEN, _NET_WM_STATE_FOCUSED") is True
|
||||
assert screen_context.parse_xprop_fullscreen("_NET_WM_STATE(ATOM) = _NET_WM_STATE_FOCUSED") is False
|
||||
|
||||
|
||||
def test_desktop_titles_are_treated_as_no_context():
|
||||
assert screen_context.clean_title("Desktop") is None
|
||||
assert screen_context.clean_title(" ") is None
|
||||
|
||||
|
||||
def test_long_titles_are_truncated():
|
||||
cleaned = screen_context.clean_title("x" * 500)
|
||||
assert len(cleaned) <= 160 and cleaned.endswith("…")
|
||||
|
||||
|
||||
def test_annotate_appends_context_as_an_aside():
|
||||
annotated = screen_context.annotate("what's this error?", "app.py — Traceback")
|
||||
assert annotated.startswith("what's this error?")
|
||||
assert "[on screen right now: app.py — Traceback]" in annotated
|
||||
|
||||
|
||||
def test_annotate_is_a_no_op_without_a_title_or_text():
|
||||
assert screen_context.annotate("hello", None) == "hello"
|
||||
assert screen_context.annotate("hello", "Desktop") == "hello"
|
||||
assert screen_context.annotate("", "Firefox") == ""
|
||||
@@ -0,0 +1,86 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import pytest
|
||||
|
||||
from bolt_pet import server_client
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _configure(monkeypatch):
|
||||
monkeypatch.setattr(server_client.config, "SERVER_URL", "http://test-server:5002")
|
||||
monkeypatch.setattr(server_client.config, "API_KEY", "test-key")
|
||||
monkeypatch.setattr(server_client.config, "SESSION_ID", "pet-test")
|
||||
|
||||
|
||||
def _mock_response(json_data, ok=True):
|
||||
resp = MagicMock()
|
||||
resp.json.return_value = json_data
|
||||
resp.raise_for_status = MagicMock() if ok else MagicMock(side_effect=Exception("boom"))
|
||||
return resp
|
||||
|
||||
|
||||
def test_converse_returns_reply_directly():
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({"type": "reply", "text": "hello there"})
|
||||
result = server_client.converse("hi")
|
||||
assert result == "hello there"
|
||||
post.assert_called_once()
|
||||
args, kwargs = post.call_args
|
||||
assert args[0] == "http://test-server:5002/desk/converse"
|
||||
assert kwargs["json"] == {"session_id": "pet-test", "text": "hi"}
|
||||
assert kwargs["headers"] == {"X-Desk-Api-Key": "test-key"}
|
||||
|
||||
|
||||
def test_converse_relays_a_command_then_returns_reply():
|
||||
responses = [
|
||||
_mock_response({"type": "command", "command": "echo hi", "token": "tok1"}),
|
||||
_mock_response({"type": "reply", "text": "done"}),
|
||||
]
|
||||
with patch.object(server_client.requests, "post", side_effect=responses) as post:
|
||||
on_command = MagicMock(return_value="[exit 0]\nhi")
|
||||
result = server_client.converse("run echo hi", on_command=on_command)
|
||||
assert result == "done"
|
||||
on_command.assert_called_once_with("echo hi")
|
||||
# second call was to /desk/tool_result with the command's output
|
||||
second_call = post.call_args_list[1]
|
||||
assert second_call.args[0] == "http://test-server:5002/desk/tool_result"
|
||||
assert second_call.kwargs["json"] == {
|
||||
"session_id": "pet-test", "token": "tok1", "output": "[exit 0]\nhi",
|
||||
}
|
||||
|
||||
|
||||
def test_converse_raises_server_error_on_error_payload():
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({"type": "error", "error": "unauthorized"})
|
||||
with pytest.raises(server_client.ServerError, match="unauthorized"):
|
||||
server_client.converse("hi")
|
||||
|
||||
|
||||
def test_converse_raises_server_error_when_unreachable():
|
||||
with patch.object(server_client.requests, "post", side_effect=ConnectionError("no route")):
|
||||
with pytest.raises(server_client.ServerError):
|
||||
server_client.converse("hi")
|
||||
|
||||
|
||||
def test_report_status_returns_reply_text_when_present():
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({"reply": "don't forget your 3pm"})
|
||||
result = server_client.report_status()
|
||||
assert result == "don't forget your 3pm"
|
||||
|
||||
|
||||
def test_report_status_returns_none_when_nothing_pending():
|
||||
with patch.object(server_client.requests, "post") as post:
|
||||
post.return_value = _mock_response({"ok": True})
|
||||
assert server_client.report_status() is None
|
||||
|
||||
|
||||
def test_check_health_returns_parsed_json():
|
||||
with patch.object(server_client.requests, "get") as get:
|
||||
get.return_value = _mock_response({"ok": True, "service": "bolt-desk-api"})
|
||||
result = server_client.check_health()
|
||||
assert result == {"ok": True, "service": "bolt-desk-api"}
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Sanitizing chat-formatted replies into speakable prose. Pure string logic
|
||||
— no audio hardware, no Qt (see the testing conventions in CLAUDE.md)."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet.speech_text import for_display, for_speech
|
||||
|
||||
|
||||
def test_bold_markers_are_not_spoken():
|
||||
assert "*" not in for_speech("Here's the **maji-desktop** snapshot")
|
||||
assert for_speech("Here's the **maji-desktop** snapshot") == "Here's the maji-desktop snapshot"
|
||||
|
||||
|
||||
def test_italics_and_underscores_dropped():
|
||||
assert for_speech("that is _really_ odd") == "that is really odd"
|
||||
assert for_speech("***everything*** is fine") == "everything is fine"
|
||||
|
||||
|
||||
def test_bullet_list_becomes_sentences():
|
||||
spoken = for_speech(
|
||||
"System status:\n"
|
||||
"* **OS:** Linux 7.0.0\n"
|
||||
"* **Uptime:** 1 day\n"
|
||||
)
|
||||
assert "*" not in spoken
|
||||
assert spoken == "System status: OS: Linux 7.0.0. Uptime: 1 day."
|
||||
|
||||
|
||||
def test_emoji_and_symbols_removed():
|
||||
assert for_speech("✅ Done 🚀 — all good") == "Done all good"
|
||||
assert for_speech("→ next step") == "to next step"
|
||||
|
||||
|
||||
def test_urls_and_code_are_not_read_out_character_by_character():
|
||||
assert for_speech("see https://example.com/x?y=1 for docs") == "see link for docs"
|
||||
assert for_speech("run `sudo reboot` now") == "run sudo reboot now"
|
||||
assert for_speech("here:\n```\nls -la\n```\n") == "here: (code)."
|
||||
|
||||
|
||||
def test_headings_quotes_and_rules_stripped():
|
||||
assert for_speech("## Summary\n---\n> quoted bit") == "Summary. quoted bit."
|
||||
|
||||
|
||||
def test_ampersand_and_percent_are_spoken_as_words():
|
||||
assert for_speech("R&D at 50% capacity") == "R and D at 50 percent capacity"
|
||||
|
||||
|
||||
def test_blank_and_symbol_only_input():
|
||||
assert for_speech("") == ""
|
||||
assert for_speech(None) == ""
|
||||
assert for_speech("***") == ""
|
||||
|
||||
|
||||
def test_display_keeps_emoji_but_drops_markdown():
|
||||
assert for_display("**Done** ✅") == "Done ✅"
|
||||
assert for_display("* one\n* two") == "• one • two"
|
||||
@@ -0,0 +1,73 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import pytest
|
||||
|
||||
from bolt_pet.state import InvalidTransition, PetState, PetStateMachine
|
||||
|
||||
|
||||
def test_starts_idle():
|
||||
sm = PetStateMachine()
|
||||
assert sm.state == PetState.IDLE
|
||||
|
||||
|
||||
def test_happy_path_transitions():
|
||||
sm = PetStateMachine()
|
||||
sm.transition(PetState.LISTENING)
|
||||
sm.transition(PetState.THINKING)
|
||||
sm.transition(PetState.TALKING)
|
||||
sm.transition(PetState.IDLE)
|
||||
assert sm.state == PetState.IDLE
|
||||
|
||||
|
||||
def test_idle_to_talking_is_allowed_for_proactive_announcements():
|
||||
# The heartbeat poll can make the pet speak unprompted (a reminder
|
||||
# firing, a nudge from the server) with no preceding listen/think leg.
|
||||
sm = PetStateMachine()
|
||||
sm.transition(PetState.TALKING)
|
||||
assert sm.state == PetState.TALKING
|
||||
|
||||
|
||||
def test_invalid_transition_raises():
|
||||
sm = PetStateMachine()
|
||||
with pytest.raises(InvalidTransition):
|
||||
sm.transition(PetState.THINKING) # can't skip straight to thinking with no utterance
|
||||
|
||||
|
||||
def test_same_state_transition_is_a_noop():
|
||||
calls = []
|
||||
sm = PetStateMachine(on_change=lambda old, new: calls.append((old, new)))
|
||||
sm.transition(PetState.IDLE) # already idle
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_on_change_callback_fires_with_old_and_new():
|
||||
calls = []
|
||||
sm = PetStateMachine(on_change=lambda old, new: calls.append((old, new)))
|
||||
sm.transition(PetState.LISTENING)
|
||||
assert calls == [(PetState.IDLE, PetState.LISTENING)]
|
||||
|
||||
|
||||
def test_every_state_can_reach_error_and_recover():
|
||||
for path in (
|
||||
[PetState.LISTENING],
|
||||
[PetState.LISTENING, PetState.THINKING],
|
||||
[PetState.LISTENING, PetState.THINKING, PetState.TALKING],
|
||||
):
|
||||
sm = PetStateMachine()
|
||||
for step in path:
|
||||
sm.transition(step)
|
||||
sm.transition(PetState.ERROR)
|
||||
sm.transition(PetState.IDLE)
|
||||
assert sm.state == PetState.IDLE
|
||||
|
||||
|
||||
def test_force_recovers_from_talking_directly_to_idle_without_validation():
|
||||
sm = PetStateMachine()
|
||||
sm.transition(PetState.LISTENING)
|
||||
sm.transition(PetState.THINKING)
|
||||
sm.transition(PetState.TALKING)
|
||||
sm.force(PetState.LISTENING) # not in TALKING's allowed set, but force skips the check
|
||||
assert sm.state == PetState.LISTENING
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Streaming-TTS chunk reassembly and the near-miss log. Both are pure —
|
||||
no network, no audio device, no ONNX model."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from bolt_pet.audio.tts import chunks_to_int16
|
||||
from bolt_pet.audio.wake_word import NearMissLog
|
||||
|
||||
|
||||
def _pcm(*values):
|
||||
return np.array(values, dtype=np.int16).tobytes()
|
||||
|
||||
|
||||
def test_whole_samples_pass_straight_through():
|
||||
chunks = list(chunks_to_int16([_pcm(1, 2), _pcm(3, 4)]))
|
||||
assert np.concatenate(chunks).tolist() == [1, 2, 3, 4]
|
||||
|
||||
|
||||
def test_a_sample_split_across_two_http_chunks_is_rejoined():
|
||||
# The killer bug this exists to prevent: an odd byte at a chunk boundary
|
||||
# shifts everything after it by one byte and plays as static.
|
||||
raw = _pcm(100, -200, 300, -400)
|
||||
chunks = list(chunks_to_int16([raw[:3], raw[3:]]))
|
||||
assert np.concatenate(chunks).tolist() == [100, -200, 300, -400]
|
||||
|
||||
|
||||
def test_many_odd_boundaries_in_a_row():
|
||||
raw = _pcm(*range(1, 21))
|
||||
pieces = [raw[i:i + 3] for i in range(0, len(raw), 3)] # every boundary odd
|
||||
assert np.concatenate(list(chunks_to_int16(pieces))).tolist() == list(range(1, 21))
|
||||
|
||||
|
||||
def test_empty_chunks_are_skipped():
|
||||
assert list(chunks_to_int16([b"", b""])) == []
|
||||
|
||||
|
||||
def test_a_dangling_byte_at_the_end_is_dropped_not_played():
|
||||
raw = _pcm(7, 8) + b"\x01"
|
||||
assert np.concatenate(list(chunks_to_int16([raw]))).tolist() == [7, 8]
|
||||
|
||||
|
||||
# ── wake-word near misses ───────────────────────────────────────────────────
|
||||
|
||||
def test_scores_just_under_the_threshold_are_recorded():
|
||||
log = NearMissLog(limit=10, margin=0.2)
|
||||
assert log.observe(0.45, threshold=0.5, timestamp=1.0) is True
|
||||
assert log.entries() == [(1.0, 0.45, 0.5)]
|
||||
|
||||
|
||||
def test_detections_and_background_noise_are_not_near_misses():
|
||||
log = NearMissLog(limit=10, margin=0.2)
|
||||
assert log.observe(0.90, threshold=0.5, timestamp=1.0) is False # it fired
|
||||
assert log.observe(0.05, threshold=0.5, timestamp=2.0) is False # just noise
|
||||
assert log.entries() == []
|
||||
|
||||
|
||||
def test_peak_tracks_every_score_not_just_near_misses():
|
||||
log = NearMissLog(limit=10, margin=0.2)
|
||||
log.observe(0.30, threshold=0.5, timestamp=1.0)
|
||||
log.observe(0.95, threshold=0.5, timestamp=2.0)
|
||||
log.observe(0.10, threshold=0.5, timestamp=3.0)
|
||||
assert log.peak == 0.95
|
||||
|
||||
|
||||
def test_the_log_is_bounded():
|
||||
log = NearMissLog(limit=3, margin=0.2)
|
||||
for i in range(10):
|
||||
log.observe(0.45, threshold=0.5, timestamp=float(i))
|
||||
assert len(log.entries()) == 3
|
||||
assert log.entries()[-1][0] == 9.0
|
||||
|
||||
|
||||
def test_clear_resets_peak_and_entries():
|
||||
log = NearMissLog(limit=3, margin=0.2)
|
||||
log.observe(0.45, threshold=0.5, timestamp=1.0)
|
||||
log.clear()
|
||||
assert log.entries() == [] and log.peak == 0.0
|
||||
@@ -0,0 +1,148 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from bolt_pet.audio import wake_word
|
||||
|
||||
|
||||
class _FakeStream:
|
||||
"""Yields a fixed sequence of frames, then silence forever."""
|
||||
|
||||
def __init__(self, frames, frame_len):
|
||||
self._frames = list(frames)
|
||||
self._frame_len = frame_len
|
||||
|
||||
def read(self, frames):
|
||||
if self._frames:
|
||||
frame = self._frames.pop(0)
|
||||
else:
|
||||
frame = np.zeros(self._frame_len, dtype=np.int16)
|
||||
return frame.reshape(-1, 1), False
|
||||
|
||||
|
||||
def _frame(frame_len):
|
||||
return np.zeros(frame_len, dtype=np.int16)
|
||||
|
||||
|
||||
class _FakeModel:
|
||||
"""Reports the given score sequence (one dict per predict() call, then
|
||||
repeats the last entry) and records reset() calls."""
|
||||
|
||||
def __init__(self, score_sequence):
|
||||
self._scores = list(score_sequence)
|
||||
self.reset_calls = 0
|
||||
|
||||
def predict(self, frame):
|
||||
if self._scores:
|
||||
return self._scores.pop(0)
|
||||
return {"thunderbolt": 0.0}
|
||||
|
||||
def reset(self):
|
||||
self.reset_calls += 1
|
||||
|
||||
|
||||
def test_returns_true_and_resets_on_detection(monkeypatch):
|
||||
frame_len = 1280
|
||||
monkeypatch.setattr(wake_word.config, "FRAME_LEN", frame_len)
|
||||
monkeypatch.setattr(wake_word.config, "SAMPLE_RATE", 16000)
|
||||
monkeypatch.setattr(wake_word.config, "WAKE_CHECK_INTERVAL_SECONDS", 0.08) # 1 frame
|
||||
monkeypatch.setattr(wake_word.config, "WAKE_WORD_THRESHOLD", 0.5)
|
||||
|
||||
stream = _FakeStream([_frame(frame_len)] * 3, frame_len)
|
||||
model = _FakeModel([{"thunderbolt": 0.1}, {"thunderbolt": 0.9}])
|
||||
|
||||
detected = wake_word.listen_for_wake_word(stream, model=model)
|
||||
|
||||
assert detected is True
|
||||
assert model.reset_calls == 1
|
||||
|
||||
|
||||
def test_returns_false_when_should_continue_goes_false_first():
|
||||
frame_len = 1280
|
||||
stream = _FakeStream([_frame(frame_len)] * 5, frame_len)
|
||||
model = _FakeModel([{"thunderbolt": 0.0}] * 5)
|
||||
calls = {"n": 0}
|
||||
|
||||
def should_continue():
|
||||
calls["n"] += 1
|
||||
return calls["n"] <= 3
|
||||
|
||||
detected = wake_word.listen_for_wake_word(
|
||||
stream, should_continue=should_continue, model=model,
|
||||
)
|
||||
assert detected is False
|
||||
assert model.reset_calls == 0
|
||||
|
||||
|
||||
def test_custom_threshold_is_respected(monkeypatch):
|
||||
frame_len = 1280
|
||||
monkeypatch.setattr(wake_word.config, "FRAME_LEN", frame_len)
|
||||
monkeypatch.setattr(wake_word.config, "SAMPLE_RATE", 16000)
|
||||
monkeypatch.setattr(wake_word.config, "WAKE_CHECK_INTERVAL_SECONDS", 0.08)
|
||||
|
||||
stream = _FakeStream([_frame(frame_len)] * 3, frame_len)
|
||||
model = _FakeModel([{"thunderbolt": 0.6}] * 3)
|
||||
calls = {"n": 0}
|
||||
|
||||
def should_continue():
|
||||
calls["n"] += 1
|
||||
return calls["n"] <= 3
|
||||
|
||||
detected = wake_word.listen_for_wake_word(
|
||||
stream, should_continue=should_continue, model=model, threshold=0.7,
|
||||
)
|
||||
assert detected is False
|
||||
|
||||
|
||||
def test_on_tick_fires_once_per_check_interval(monkeypatch):
|
||||
frame_len = 1280
|
||||
monkeypatch.setattr(wake_word.config, "FRAME_LEN", frame_len)
|
||||
monkeypatch.setattr(wake_word.config, "SAMPLE_RATE", 16000)
|
||||
monkeypatch.setattr(wake_word.config, "WAKE_CHECK_INTERVAL_SECONDS", 0.08) # 1 frame/check
|
||||
|
||||
frames = [_frame(frame_len) for _ in range(5)]
|
||||
stream = _FakeStream(frames, frame_len)
|
||||
model = _FakeModel([{"thunderbolt": 0.0}] * 5)
|
||||
ticks = {"n": 0}
|
||||
state = {"i": 0}
|
||||
|
||||
def should_continue():
|
||||
state["i"] += 1
|
||||
return state["i"] <= 5
|
||||
|
||||
wake_word.listen_for_wake_word(
|
||||
stream,
|
||||
should_continue=should_continue,
|
||||
model=model,
|
||||
on_tick=lambda: ticks.__setitem__("n", ticks["n"] + 1),
|
||||
)
|
||||
assert ticks["n"] == 5 # one check-interval per frame at this config
|
||||
|
||||
|
||||
def test_on_tick_interval_can_span_multiple_frames(monkeypatch):
|
||||
frame_len = 1280
|
||||
monkeypatch.setattr(wake_word.config, "FRAME_LEN", frame_len)
|
||||
monkeypatch.setattr(wake_word.config, "SAMPLE_RATE", 16000)
|
||||
monkeypatch.setattr(wake_word.config, "WAKE_CHECK_INTERVAL_SECONDS", 0.24) # 3 frames/check
|
||||
|
||||
frames = [_frame(frame_len) for _ in range(6)]
|
||||
stream = _FakeStream(frames, frame_len)
|
||||
model = _FakeModel([{"thunderbolt": 0.0}] * 6)
|
||||
ticks = {"n": 0}
|
||||
state = {"i": 0}
|
||||
|
||||
def should_continue():
|
||||
state["i"] += 1
|
||||
return state["i"] <= 6
|
||||
|
||||
wake_word.listen_for_wake_word(
|
||||
stream,
|
||||
should_continue=should_continue,
|
||||
model=model,
|
||||
on_tick=lambda: ticks.__setitem__("n", ticks["n"] + 1),
|
||||
)
|
||||
assert ticks["n"] == 2 # 6 frames / 3 frames-per-check
|
||||
@@ -0,0 +1,89 @@
|
||||
"""Autonomous wandering. Needs a QApplication, so run with
|
||||
QT_QPA_PLATFORM=offscreen (same as test_controller.py)."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from PySide6.QtWidgets import QApplication
|
||||
|
||||
from bolt_pet import config
|
||||
from bolt_pet.state import PetState
|
||||
from bolt_pet.ui.pet_window import PetWindow
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def qt_app():
|
||||
app = QApplication.instance() or QApplication([])
|
||||
yield app
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pet(qt_app):
|
||||
window = PetWindow()
|
||||
window.set_wander_enabled(True)
|
||||
yield window
|
||||
window.close()
|
||||
|
||||
|
||||
def _walk_until_done(pet, max_ticks=2000):
|
||||
for _ in range(max_ticks):
|
||||
pet._wander_tick()
|
||||
if pet._wander_target is None:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def test_wanders_when_idle_and_reaches_its_target(pet, monkeypatch):
|
||||
# Snapping is tested separately; here it would tug the pet off its target
|
||||
# the moment it arrives near an edge.
|
||||
monkeypatch.setattr(config, "PET_EDGE_SNAP", False)
|
||||
pet.wander_now()
|
||||
pet._wander_tick()
|
||||
target = pet._wander_target
|
||||
if target is None:
|
||||
pytest.skip("no usable screen geometry for a stroll on this host")
|
||||
assert _walk_until_done(pet), "pet never reached its wander target"
|
||||
assert pet.pos() == target
|
||||
|
||||
|
||||
def test_stays_put_while_talking(pet):
|
||||
pet.set_state(PetState.TALKING)
|
||||
start = pet.pos()
|
||||
pet.wander_now()
|
||||
for _ in range(50):
|
||||
pet._wander_tick()
|
||||
assert pet.pos() == start
|
||||
assert pet._wander_target is None
|
||||
|
||||
|
||||
def test_stays_put_while_bubble_is_up(pet):
|
||||
pet.say("hello there", duration_ms=60000)
|
||||
start = pet.pos()
|
||||
pet.wander_now()
|
||||
for _ in range(50):
|
||||
pet._wander_tick()
|
||||
assert pet.pos() == start
|
||||
|
||||
|
||||
def test_disabling_wander_stops_movement(pet):
|
||||
pet.set_wander_enabled(False)
|
||||
start = pet.pos()
|
||||
pet.wander_now()
|
||||
for _ in range(50):
|
||||
pet._wander_tick()
|
||||
assert pet.pos() == start
|
||||
|
||||
|
||||
def test_stroll_stays_on_screen(pet):
|
||||
geo = pet._screen_geometry()
|
||||
if geo is None:
|
||||
pytest.skip("no usable screen geometry on this host")
|
||||
for _ in range(10):
|
||||
pet.wander_now()
|
||||
pet._wander_tick()
|
||||
assert _walk_until_done(pet)
|
||||
assert geo.contains(pet.geometry())
|
||||