Add text-to-dialogue, self-restart capability, and misc updates
This commit is contained in:
@@ -0,0 +1,219 @@
|
||||
"""`dialoguectl` — multi-voice dialogue playback (ElevenLabs Text to Dialogue).
|
||||
|
||||
Normal replies are one voice saying one thing (audio/tts.py). This is the
|
||||
other mode: a short *scene* — two or more voices, with delivery tags the v3
|
||||
model acts on (`[cheerfully]`, `[stuttering]`, `[whispering]`) — synthesized
|
||||
as a single take so the timing and reactions between lines actually sound
|
||||
like a conversation rather than clips glued together.
|
||||
|
||||
Wire format, the same discipline as file_ops.py and for the same reason: it
|
||||
rides the server's ordinary `command` tool marker, whose extractor only
|
||||
captures up to the next newline, so the payload is a **single-line compact
|
||||
JSON object**.
|
||||
|
||||
dialoguectl {"lines": [{"voice": "self", "text": "[cheerfully] Morning!"},
|
||||
{"voice": "narrator", "text": "[whispering] He lies."}]}
|
||||
|
||||
The ElevenLabs field names are accepted too (`inputs` / `voice_id`), because
|
||||
the model has read that API and copying its shape is the obvious thing to
|
||||
try:
|
||||
|
||||
dialoguectl {"inputs": [{"voice_id": "9BWtsMINqrJLrRacOk9x", "text": "hi"}]}
|
||||
|
||||
Voices are *named*, not pasted as ids. `DIALOGUE_VOICES` in .env maps names
|
||||
to ids (`narrator:9BWts…,villain:IKne3…`), and `self` always means the voice
|
||||
the pet is speaking with right now — including a voice the server picked
|
||||
mid-conversation with `speak_as`, so a scene featuring Bolt sounds like
|
||||
whoever Bolt currently is.
|
||||
|
||||
Pure parsing and validation here; the HTTP call is
|
||||
`audio/tts.synthesize_dialogue` and the playback/state handling is
|
||||
`controller._play_dialogue`, matching the parse/execute split used by
|
||||
pet_actions.py and file_ops.py.
|
||||
|
||||
The API's own limits are enforced *here*, before the request goes out, so a
|
||||
mistake comes back through the tool-result relay as a sentence Bolt can act
|
||||
on ("too many characters, split it") rather than as an HTTP 422 he can't see.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from typing import Iterable, Optional
|
||||
|
||||
_PREFIXES = ("dialoguectl", "dialogue", "scene")
|
||||
|
||||
# ElevenLabs Text to Dialogue limits (docs, 2026-07): at most 10 distinct
|
||||
# voice ids per request and ~2000 characters across all inputs.
|
||||
MAX_VOICES = 10
|
||||
MAX_CHARS = 2000
|
||||
|
||||
# Names that always mean "the voice the pet is using right now".
|
||||
SELF_NAMES = ("self", "bolt", "me", "pet")
|
||||
|
||||
# A raw ElevenLabs voice id: 20 URL-safe characters, no separators. Used to
|
||||
# tell "the model pasted an id" from "the model used a name".
|
||||
_VOICE_ID_RE = re.compile(r"^[A-Za-z0-9]{20}$")
|
||||
|
||||
|
||||
class DialogueError(Exception):
|
||||
"""Bad dialoguectl syntax or an unusable request — reported back to the
|
||||
server as this command's output."""
|
||||
|
||||
|
||||
def is_dialogue_command(command: str) -> bool:
|
||||
parts = (command or "").strip().split(None, 1)
|
||||
return bool(parts) and parts[0].lower() in _PREFIXES
|
||||
|
||||
|
||||
def parse(command: str) -> Optional[dict]:
|
||||
"""Parse `dialoguectl <json>` into {"lines": [{"voice", "text"}], ...}.
|
||||
|
||||
Returns None if this isn't a dialogue command at all (the caller then
|
||||
tries filectl, then a real shell command). Raises DialogueError on a
|
||||
dialogue command that doesn't make sense."""
|
||||
if not is_dialogue_command(command):
|
||||
return None
|
||||
_, _, payload = (command or "").strip().partition(" ")
|
||||
payload = payload.strip()
|
||||
if not payload:
|
||||
raise DialogueError(
|
||||
'dialoguectl needs a JSON argument, e.g. dialoguectl {"lines": '
|
||||
'[{"voice": "self", "text": "[cheerfully] hello"}]}'
|
||||
)
|
||||
try:
|
||||
data = json.loads(payload)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise DialogueError(
|
||||
f"couldn't parse the JSON ({exc}). It must be one line of compact "
|
||||
"JSON — put line breaks inside text as \\n, never as real newlines."
|
||||
) from exc
|
||||
if not isinstance(data, dict):
|
||||
raise DialogueError("the argument must be a JSON object, not a list or a bare value")
|
||||
|
||||
raw_lines = data.get("lines")
|
||||
if raw_lines is None:
|
||||
raw_lines = data.get("inputs") # the ElevenLabs field name
|
||||
if not isinstance(raw_lines, list) or not raw_lines:
|
||||
raise DialogueError('needs a non-empty "lines" array of {"voice", "text"} objects')
|
||||
|
||||
lines: list[dict] = []
|
||||
for index, entry in enumerate(raw_lines, start=1):
|
||||
if not isinstance(entry, dict):
|
||||
raise DialogueError(f"line {index} must be an object with 'voice' and 'text'")
|
||||
text = str(entry.get("text") or "").strip()
|
||||
if not text:
|
||||
raise DialogueError(f"line {index} has no text")
|
||||
voice = str(entry.get("voice") or entry.get("voice_id") or "self").strip()
|
||||
lines.append({"voice": voice, "text": text})
|
||||
|
||||
action = {"action": "dialogue", "lines": lines}
|
||||
model = str(data.get("model") or data.get("model_id") or "").strip()
|
||||
if model:
|
||||
action["model"] = model
|
||||
stability = data.get("stability")
|
||||
if stability is not None:
|
||||
try:
|
||||
action["stability"] = min(1.0, max(0.0, float(stability)))
|
||||
except (TypeError, ValueError):
|
||||
raise DialogueError("stability must be a number between 0 and 1") from None
|
||||
return action
|
||||
|
||||
|
||||
def parse_voice_map(spec: str) -> dict[str, str]:
|
||||
"""Parse DIALOGUE_VOICES ("narrator:9BWts…, villain:IKne3…") into a map.
|
||||
|
||||
Malformed entries are skipped rather than raising: a typo in .env should
|
||||
cost that one voice, not the whole feature."""
|
||||
voices: dict[str, str] = {}
|
||||
for chunk in str(spec or "").split(","):
|
||||
name, separator, voice_id = chunk.partition(":")
|
||||
name, voice_id = name.strip().lower(), voice_id.strip()
|
||||
if separator and name and voice_id:
|
||||
voices[name] = voice_id
|
||||
return voices
|
||||
|
||||
|
||||
def resolve(
|
||||
action: dict,
|
||||
*,
|
||||
voices: Optional[dict] = None,
|
||||
self_voice: str = "",
|
||||
) -> list[dict]:
|
||||
"""Turn parsed lines into the API's `inputs`, resolving names to ids.
|
||||
|
||||
*self_voice* is the pet's current voice (which may be a `speak_as` pick,
|
||||
not the configured default), so "self" tracks whoever Bolt sounds like
|
||||
right now."""
|
||||
known = dict(voices or {})
|
||||
resolved: list[dict] = []
|
||||
for index, line in enumerate(action.get("lines") or [], start=1):
|
||||
name = str(line.get("voice") or "self")
|
||||
key = name.lower()
|
||||
if key in SELF_NAMES:
|
||||
voice_id = self_voice
|
||||
if not voice_id:
|
||||
raise DialogueError(
|
||||
"no voice is configured for the pet itself — set "
|
||||
"ELEVENLABS_VOICE_ID, or name a voice from DIALOGUE_VOICES"
|
||||
)
|
||||
elif key in known:
|
||||
voice_id = known[key]
|
||||
elif _VOICE_ID_RE.match(name):
|
||||
voice_id = name # a raw id pasted straight from the voice library
|
||||
else:
|
||||
available = ", ".join(sorted(known) + list(SELF_NAMES[:1])) or "self"
|
||||
raise DialogueError(
|
||||
f"line {index}: unknown voice {name!r}. Known names: {available}. "
|
||||
"Use one of those, 'self' for your own voice, or a raw voice id."
|
||||
)
|
||||
resolved.append({"text": str(line.get("text") or ""), "voice_id": voice_id})
|
||||
|
||||
check_limits(resolved)
|
||||
return resolved
|
||||
|
||||
|
||||
def check_limits(inputs: Iterable[dict], *, max_voices: int = MAX_VOICES,
|
||||
max_chars: int = MAX_CHARS) -> None:
|
||||
"""Enforce the API's own limits before spending a request on a 422."""
|
||||
entries = list(inputs)
|
||||
if not entries:
|
||||
raise DialogueError("no lines to speak")
|
||||
distinct = {entry["voice_id"] for entry in entries}
|
||||
if len(distinct) > max_voices:
|
||||
raise DialogueError(
|
||||
f"{len(distinct)} different voices — the limit is {max_voices} per scene"
|
||||
)
|
||||
total = sum(len(entry["text"]) for entry in entries)
|
||||
if total > max_chars:
|
||||
raise DialogueError(
|
||||
f"{total} characters — the limit is {max_chars} per scene. "
|
||||
"Split it into two dialoguectl calls."
|
||||
)
|
||||
|
||||
|
||||
def spoken_text(action: dict) -> str:
|
||||
"""The scene as readable text, for the speech bubble and the transcript.
|
||||
|
||||
Delivery tags are stripped: `[cheerfully]` is a stage direction for the
|
||||
model, not something to show (or, via tts.speak's sanitizer, to read out)."""
|
||||
parts = []
|
||||
for line in action.get("lines") or []:
|
||||
text = re.sub(r"\[[^\]]{1,40}\]", " ", str(line.get("text") or ""))
|
||||
text = " ".join(text.split())
|
||||
if text:
|
||||
parts.append(text)
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def describe(action: dict, *, played: bool = True) -> str:
|
||||
"""The tool-result string handed back to the server."""
|
||||
lines = action.get("lines") or []
|
||||
voices = sorted({str(line.get("voice") or "self") for line in lines})
|
||||
if not played:
|
||||
return f"[dialogue] not played ({len(lines)} lines)"
|
||||
return (
|
||||
f"[dialogue] played {len(lines)} line{'s' if len(lines) != 1 else ''} "
|
||||
f"in {len(voices)} voice{'s' if len(voices) != 1 else ''}: {', '.join(voices)}"
|
||||
)
|
||||
Reference in New Issue
Block a user