diff --git a/flash/ai.py b/flash/ai.py index 6dd886e..5a3eb1f 100644 --- a/flash/ai.py +++ b/flash/ai.py @@ -157,12 +157,12 @@ === Voice Mode === The user is speaking to you, and your reply is read back to them out loud. -Keep it short and plain: whole sentences, no code blocks, tables, or long -lists unless they ask for one, because only the prose is spoken and the -rest is silently dropped. What they said reached you through speech -recognition, so expect missing punctuation and the occasional misheard -word; ask when a name, path, or command sounds wrong rather than acting on -a guess.""".rstrip() +Keep it short: two or three sentences unless they ask for more. Plain whole +sentences, no code blocks, tables, lists, or emojis unless they ask for +one, because only the prose is spoken and the rest is silently dropped. +What they said reached you through speech recognition, so expect missing +punctuation and the occasional misheard word; ask when a name, path, or +command sounds wrong rather than acting on a guess.""".rstrip() class FlashError(Exception): diff --git a/flash/version.py b/flash/version.py index ead868c..942c51a 100644 --- a/flash/version.py +++ b/flash/version.py @@ -1,4 +1,4 @@ -__version__ = "0.5.10" +__version__ = "0.5.11" REPO = "Natuworkguy/Flash" REPO_URL = f"https://github.com/{REPO}" diff --git a/flash/voice.py b/flash/voice.py index d1ac8b3..7842ff5 100644 --- a/flash/voice.py +++ b/flash/voice.py @@ -7,14 +7,20 @@ """ import array +import io import json +import logging import math import os import re +import shutil import sys import threading +import unicodedata +import wave import zipfile from collections.abc import Callable +from dataclasses import dataclass from pathlib import Path from urllib.error import URLError from urllib.request import Request, urlopen @@ -133,6 +139,36 @@ def interrupt_word() -> str: MAX_TURN_SECONDS = 120.0 + +@dataclass(frozen=True) +class Choice: + """A model Settings offers: its name, download size, and who it suits.""" + + name: str + size: int + label: str + + +# What Settings offers to listen with, lightest first. Sizes are the +# downloads, measured from the servers that host them. +LISTENING = ( + Choice(DEFAULT_VOSK_MODEL, 41_205_931, + "Small and quick, for weaker machines"), + Choice("vosk-model-en-us-0.22-lgraph", 130_557_655, + "Balanced, noticeably more accurate"), + Choice("vosk-model-en-us-0.22", 1_913_365_522, + "Most accurate, for strong machines"), +) + +# And to speak with. +SPEAKING = ( + Choice("en_US-amy-low", 63_104_526, "Amy, fastest, for weaker machines"), + Choice(DEFAULT_PIPER_VOICE, 63_201_294, "Amy, balanced"), + Choice("en_US-lessac-high", 113_895_201, "Lessac, clearest"), + Choice("en_US-ryan-high", 120_786_792, + "Ryan, clearest, for strong machines"), +) + DOWNLOAD_TIMEOUT = 30 DOWNLOAD_CHUNK = 1 << 16 @@ -140,21 +176,30 @@ def interrupt_word() -> str: State = Callable[[str], None] -def vosk_model_dir() -> Path: - """Where the unpacked Vosk model lives once downloaded.""" +def vosk_model_dir(name: str = "") -> Path: + """Where an unpacked Vosk model lives once downloaded.""" - return MODELS_DIR / vosk_model() + return MODELS_DIR / (name or vosk_model()) -def piper_paths() -> tuple[Path, Path]: - """The Piper voice's network and its config file.""" +def piper_paths(name: str = "") -> tuple[Path, Path]: + """A Piper voice's network and its config file.""" - onnx = MODELS_DIR / f"{piper_voice()}.onnx" + onnx = MODELS_DIR / f"{name or piper_voice()}.onnx" return onnx, onnx.with_suffix(".onnx.json") -def _piper_urls() -> tuple[str, str]: +def listening_installed(name: str) -> bool: + return (vosk_model_dir(name) / "am").is_dir() + + +def voice_installed(name: str) -> bool: + onnx, config = piper_paths(name) + return onnx.is_file() and config.is_file() + + +def _piper_urls(name: str = "") -> tuple[str, str]: """The download addresses for the configured Piper voice. A voice is named locale-speaker-quality ("en_US-amy-medium"), and @@ -163,7 +208,7 @@ def _piper_urls() -> tuple[str, str]: name that is not shaped like a voice. """ - name = piper_voice() + name = name or piper_voice() locale, speaker, quality = name.split("-", 2) language = locale.split("_")[0].lower() base = f"{PIPER_BASE}/{language}/{locale}/{speaker}/{quality}/{name}.onnx" @@ -203,6 +248,16 @@ def missing_packages() -> list[str]: return missing +def web_missing() -> list[str]: + """What voice in the web UI still needs installed. + + Listening and speaking only: the browser has the microphone and the + speakers, so sounddevice is the terminal's business, not the page's. + """ + + return [p for p in missing_packages() if p != "sounddevice"] + + def _download(url: str, out: Path, label: str, on_progress: Progress) -> str: """Stream `url` to `out`, reporting percent complete as it goes.""" @@ -266,51 +321,72 @@ def ensure_models(on_progress: Progress) -> str: if models_present(): return "" + return (download_listening(vosk_model(), on_progress) + or download_voice(piper_voice(), on_progress)) + + +def download_listening(name: str, on_progress: Progress) -> str: + """Fetch and unpack the Vosk model NAME. Returns "" once it is in.""" + + if listening_installed(name): + return "" + MODELS_DIR.mkdir(parents=True, exist_ok=True) + archive = MODELS_DIR / f"{name}.zip" + why = _download( + f"{VOSK_BASE}/{name}.zip", archive, "listening model", on_progress, + ) + if why: + return why + + why = _unpack(archive, MODELS_DIR) + if why: + return why - if not (vosk_model_dir() / "am").is_dir(): - name = vosk_model() - archive = MODELS_DIR / f"{name}.zip" - why = _download( - f"{VOSK_BASE}/{name}.zip", - archive, - "listening model", - on_progress, + if not listening_installed(name): + return ( + f"{name} did not unpack into {vosk_model_dir(name)}. Check the " + "model name in VOICE_VOSK_MODEL." ) - if why: - return why + return "" - why = _unpack(archive, MODELS_DIR) - if why: - return why - if not (vosk_model_dir() / "am").is_dir(): - return ( - f"{name} did not unpack into " - f"{vosk_model_dir()}. Check the model name in " - "VOICE_VOSK_MODEL." - ) +def download_voice(name: str, on_progress: Progress) -> str: + """Fetch the Piper voice NAME, network and settings. Returns "" once + it is in.""" - onnx, config = piper_paths() + if voice_installed(name): + return "" - if not onnx.is_file() or not config.is_file(): - try: - onnx_url, config_url = _piper_urls() - except ValueError: - return ( - f"'{piper_voice()}' is not a Piper voice name. Use " - "one shaped like en_US-amy-medium." - ) + try: + onnx_url, config_url = _piper_urls(name) + except ValueError: + return ( + f"'{name}' is not a Piper voice name. Use one shaped like " + "en_US-amy-medium." + ) - why = _download(onnx_url, onnx, "voice", on_progress) - if why: - return why + onnx, config = piper_paths(name) + MODELS_DIR.mkdir(parents=True, exist_ok=True) + return (_download(onnx_url, onnx, "voice", on_progress) + or _download(config_url, config, "voice settings", on_progress)) - why = _download(config_url, config, "voice settings", on_progress) - if why: - return why - return "" +def remove_listening(name: str) -> None: + """Delete the Vosk model NAME, which must be one Settings offers.""" + + if name not in {c.name for c in LISTENING}: + raise ValueError(f"{name!r} is not a listening model Flash offers") + shutil.rmtree(vosk_model_dir(name), ignore_errors=True) + + +def remove_voice(name: str) -> None: + """Delete the Piper voice NAME, which must be one Settings offers.""" + + if name not in {c.name for c in SPEAKING}: + raise ValueError(f"{name!r} is not a voice Flash offers") + for path in piper_paths(name): + path.unlink(missing_ok=True) _listener = None @@ -568,6 +644,10 @@ def _load_speaker(): if _speaker is None or _speaker[0] != onnx: from piper import PiperVoice + # A sound the voice lacks is dropped either way; Piper saying so + # in the terminal, once per sound, only gets in the way. + logging.getLogger("piper").setLevel(logging.ERROR) + settings = json.loads(config.read_text(encoding="utf-8")) rate = int(settings.get("audio", {}).get("sample_rate", 22050)) _speaker = (onnx, PiperVoice.load(str(onnx)), rate) @@ -575,9 +655,49 @@ def _load_speaker(): return _speaker[1], _speaker[2] +# Symbols a voice cannot say as written, with the words it can. Longer +# ones first, so "->" is not read as "-" then ">". +_SAID_AS = ( + ("->", " to "), ("=>", " to "), ("\u2192", " to "), ("\u21d2", " to "), + (">=", " at least "), ("\u2265", " at least "), + ("<=", " at most "), ("\u2264", " at most "), + ("!=", " is not "), ("\u2260", " is not "), ("==", " equals "), + ("&&", " and "), ("||", " or "), ("&", " and "), + ("\u00b1", " plus or minus "), ("\u00d7", " times "), + ("\u00f7", " divided by "), ("\u00b0", " degrees "), + ("\u2044", " over "), + ("+", " plus "), ("=", " equals "), ("@", " at "), +) +# Marks from code and Markdown, said as a pause, not spelled out. +# Brackets above all: Piper takes [[ ... ]] as raw phonemes, and reads +# whatever is inside as sounds its voice may not have. +_UNSAID = re.compile(r"[\[\]{}<>|`*#^~_\\\u2022\u2023\u25e6]+") + + +def fit_for_voice(text: str) -> str: + """TEXT in words a voice can say. + + Code, paths, and symbols come through otherwise as sounds the voice + has none of: Piper drops them with a warning, and what is left comes + out garbled. + """ + + # Compatibility forms first: ² as 2, fi as fi, a full-width letter + # as itself. + text = unicodedata.normalize("NFKC", text or "") + for mark, words in _SAID_AS: + text = text.replace(mark, words) + text = _UNSAID.sub(" ", text) + return " ".join(text.split()) + + def _pcm_chunks(voice, text: str): """Yield raw 16-bit audio for `text`, across Piper's two APIs.""" + text = fit_for_voice(text) + if not text: + return + if hasattr(voice, "synthesize_stream_raw"): yield from voice.synthesize_stream_raw(text) return @@ -655,6 +775,105 @@ def _abort(stream) -> None: pass +# The page asks for both from threads of its own; loading a model twice +# at once would only waste the memory of one, and Piper is not promised +# to be safe to call from two threads together. +_models_lock = threading.Lock() + + +def transcribe(pcm: bytes) -> tuple[str, str]: + """What was said in PCM: 16 kHz mono 16-bit audio, as the web page + records it. Returns `(text, "")`, or `("", reason)` when the listening + model cannot be used.""" + + try: + import vosk + except Exception: # noqa: BLE001 + return "", INSTALL_HINT + + try: + with _models_lock: + model = _load_listener() + except Exception as exc: # noqa: BLE001 + return "", f"could not load the listening model: {exc}" + + recognizer = vosk.KaldiRecognizer(model, SAMPLE_RATE) + block = BLOCK_FRAMES * 2 + for start in range(0, len(pcm) - len(pcm) % 2, block): + recognizer.AcceptWaveform(pcm[start:start + block]) + + try: + result = json.loads(recognizer.FinalResult()) + except ValueError: + return "", "the listening model returned nothing usable." + + return str(result.get("text") or "").strip(), "" + + +def synthesize(text: str) -> tuple[bytes, str]: + """TEXT spoken, as a WAV file the web page can play. Returns + `(wav, "")`, or `(b"", reason)` when the voice cannot be used.""" + + text = (text or "").strip() + if not text: + return b"", "" + + try: + with _models_lock: + voice, rate = _load_speaker() + pcm = b"".join(chunk or b"" for chunk in _pcm_chunks(voice, text)) + except ImportError: + return b"", INSTALL_HINT + except Exception as exc: # noqa: BLE001 + return b"", f"could not speak the reply: {exc}" + + out = io.BytesIO() + with wave.open(out, "wb") as wav: + wav.setnchannels(1) + wav.setsampwidth(2) + wav.setframerate(rate) + wav.writeframes(pcm) + return out.getvalue(), "" + + +def speakable(line: str) -> str: + """One line of a reply, as the page reads it off the screen, made fit + to hear: an address is said as "a link", not spelled out, and emojis + are left unsaid.""" + + return " ".join(_EMOJI.sub("", _URL.sub("a link", line or "")).split()) + + +# Said to a paused session, these wake it; said as a whole turn, these +# pause it. Matched on the words alone, like the exit phrases. +RESUME_WORD = "resume" +PAUSE_PHRASES = {"pause", "pause voice", "pause voice mode", "hold on"} + + +def _words(text: str) -> list[str]: + return re.sub(r"[^a-z' ]", "", (text or "").lower()).split() + + +def voice_command(text: str, speaking: str = "") -> dict: + """What a turn of speech asks of voice mode, beyond being a message. + + `speaking` is the line Flash was reading out when it was heard: said + over a reply, "interrupt" cuts it off, but not when the word came + from the reply itself, echoing back through the microphone. + """ + + words = _words(text) + return { + "exit": is_exit_phrase(text), + "pause": " ".join(words) in PAUSE_PHRASES, + "resume": RESUME_WORD in words, + "interrupt": _heard_interruption(text) + and not _heard_interruption(speaking), + # Only the interrupt word, with nothing playing to cut off. + "alone": " ".join(words) == interrupt_word(), + } + + CODE_ONLY = "That reply is code. It is on screen." CUT_SHORT = "The rest is on screen." @@ -662,6 +881,12 @@ def _abort(stream) -> None: _IMAGE = re.compile(r"!\[[^\]]*\]\([^)]*\)") _LINK = re.compile(r"\[([^\]]+)\]\([^)]*\)") _URL = re.compile(r"?") +# Pictographs, dingbats, flags, and the joiners and selectors that knit +# them together: seen on screen, never read out. +_EMOJI = re.compile( + "[\U0001F000-\U0001FAFF\u2600-\u27BF\u2B00-\u2BFF" + "\uFE0E\uFE0F\u200D\u20E3\U000E0020-\U000E007F]" +) _TABLE = re.compile(r"^\s*\|.*$", re.MULTILINE) _RULE = re.compile(r"^\s*([-*_]\s*){3,}$", re.MULTILINE) _HEADING = re.compile(r"^\s{0,3}#{1,6}\s*", re.MULTILINE) @@ -685,7 +910,7 @@ def for_speech(text: str, limit: int = 0) -> str: # the reader can finish on screen. limit = limit or max_speech_chars() - stripped = _FENCE.sub(" ", text or "") + stripped = _EMOJI.sub("", _FENCE.sub(" ", text or "")) stripped = _IMAGE.sub(" ", stripped) stripped = _LINK.sub(r"\1", stripped) stripped = _URL.sub("a link", stripped) diff --git a/flash/web.py b/flash/web.py index fc696e2..7aba5f6 100644 --- a/flash/web.py +++ b/flash/web.py @@ -57,6 +57,7 @@ memory, skills, updater, + voice, workspace, ) from .dashes import DashGuard @@ -118,6 +119,11 @@ MAX_BODY_BYTES = 1_000_000 # An upload arrives as base64 in JSON: a third bigger than the file. MAX_UPLOAD_BODY = workspace.MAX_UPLOAD_BYTES * 4 // 3 + 64_000 +# Voice from the page arrives as base64 16 kHz 16-bit mono audio, up to +# the longest turn the terminal's voice mode would record. +MAX_VOICE_BODY = ( + int(voice.MAX_TURN_SECONDS) * voice.SAMPLE_RATE * 2 * 4 // 3 + 64_000 +) # The most attachments one message carries. MAX_ATTACHMENTS = 10 @@ -238,6 +244,9 @@ class Chat: # running turn's next step; a "queue" one becomes the next turn. pending: list[dict] = field(default_factory=list) stop: threading.Event = field(default_factory=threading.Event) + # The turn running now came from voice mode: its reply is heard, so + # the model is asked to keep it short and plain. + heard: bool = False created: float = field(default_factory=time.time) updated: float = field(default_factory=time.time) project: str = "" @@ -621,6 +630,10 @@ class Session: def __init__(self, hub: Optional[Hub] = None) -> None: self.hub = hub or Hub() self.lan = False + # The voice models are being downloaded for the page: the first + # use's pair, or one model picked in Settings, (kind, name). + self.voice_setup = False + self.voice_job: Optional[tuple[str, str]] = None # The link's token and the browsers signed in with it. A server # restarted after an update takes over the old one's (Server). self.access = Access() @@ -732,6 +745,82 @@ def state(self, lite: bool = False) -> dict: }, } + # Voice --------------------------------------------------------- + + def set_up_voice(self) -> None: + """Download the voice models, once, telling every page how far + along it is.""" + + with self._lock: + if self.voice_setup or self.voice_job: + return + self.voice_setup = True + + def run() -> None: + said: dict[str, int] = {} + + def progress(label: str, percent: int) -> None: + if said.get(label) != percent: + said[label] = percent + self.hub.publish({ + "type": "voice-setup", "label": label, + "percent": percent, + }) + + why = "" + try: + why = voice.ensure_models(progress) + finally: + with self._lock: + self.voice_setup = False + self.hub.publish( + {"type": "voice-setup", "done": True, "error": why} + ) + + threading.Thread(target=run, daemon=True).start() + + def fetch_voice_model(self, kind: str, name: str) -> None: + """Download one voice model picked in Settings and, once it is + in, make it the one in use. Pages follow along through + "voice-model" events.""" + + from . import ai # deferred: ai imports half of Flash + + with self._lock: + if self.voice_setup or self.voice_job: + raise ValueError("A voice model is already downloading.") + self.voice_job = (kind, name) + + def run() -> None: + said: list[int] = [-1] + + def progress(label: str, percent: int) -> None: + if said[0] != percent: + said[0] = percent + self.hub.publish({ + "type": "voice-model", "kind": kind, "name": name, + "label": label, "percent": percent, + }) + + why = "" + try: + fetch = (voice.download_listening if kind == "listening" + else voice.download_voice) + why = fetch(name, progress) + if not why: + ai.set_config_var(VOICE_SETTINGS[kind], name) + except Exception as exc: # noqa: BLE001 + why = str(exc) + finally: + with self._lock: + self.voice_job = None + self.hub.publish({ + "type": "voice-model", "kind": kind, "name": name, + "done": True, "error": why, + }) + + threading.Thread(target=run, daemon=True).start() + # Sub-agents ---------------------------------------------------- def adopt(self, chat: Chat, agent_ids: set) -> None: @@ -903,7 +992,7 @@ def answer(self, ask_id: str, answer: str) -> bool: def send( self, chat: Chat, text: str, wake: bool = False, mode: str = "", - files: Optional[list] = None, + files: Optional[list] = None, heard: bool = False, ) -> dict: """Start a turn on a thread of its own and return at once. @@ -934,6 +1023,7 @@ def send( else: chat.stop.clear() chat.queued = True + chat.heard = heard waiting = False if waiting: @@ -1314,7 +1404,7 @@ def run_turn( with capture_tool_output(_sink(session, chat)), \ answer_from(session.answerer(chat)): ai._fit_and_compact(ai.console, client, chat.messages) - prompt = ai._session_system_prompt() + prompt = ai._session_system_prompt(heard=chat.heard) if found is not None: prompt = f"{prompt}\n\n{project_prompt(found)}".strip() system = ai._message("system", prompt) @@ -1627,6 +1717,44 @@ def scene_data(name: str) -> Optional[dict]: } +# The two kinds of voice model Settings manages: which catalogue, how to +# tell one is in, which setting names the one in use, how to remove one. +VOICE_KINDS = { + "listening": ("LISTENING", "listening_installed", "vosk_model", + "remove_listening"), + "speaking": ("SPEAKING", "voice_installed", "piper_voice", + "remove_voice"), +} +VOICE_SETTINGS = { + "listening": "VOICE_VOSK_MODEL", "speaking": "VOICE_PIPER_VOICE", +} + + +def voice_models(session: "Session") -> dict: + """Every voice model Settings offers, and where each one stands.""" + + def listed(kind: str) -> list[dict]: + offered, installed, setting, _ = VOICE_KINDS[kind] + in_use = getattr(voice, setting)() + return [ + { + "name": c.name, "size": c.size, "label": c.label, + "installed": getattr(voice, installed)(c.name), + "current": c.name == in_use, + } + for c in getattr(voice, offered) + ] + + job = session.voice_job + return { + "listening": listed("listening"), + "speaking": listed("speaking"), + "downloading": {"kind": job[0], "name": job[1]} if job else None, + "missing": voice.web_missing(), + "install": voice.INSTALL_HINT, + } + + def _extension_info(ext: "extensions.Extension") -> dict: return { "name": ext.name, @@ -1867,6 +1995,42 @@ def command(session: Session, body: dict, browser: str = "") -> dict: session.relink() return {"signed_out": len(gone), "you": bool(browser in gone)} + if name == "voice-status": + return { + "missing": voice.web_missing(), + "ready": voice.models_present(), + "install": voice.INSTALL_HINT, + } + + if name == "voice-setup": + # The models download once, on a thread; the page follows along + # through "voice-setup" events. + if voice.models_present(): + return {"ready": True} + session.set_up_voice() + return {"ready": False} + + if name == "voice-models": + return voice_models(session) + + if name in ("voice-model", "voice-model-remove"): + kind = str(body.get("kind") or "") + if kind not in VOICE_KINDS: + raise ValueError("pick listening or speaking") + offered, installed, setting, remove = VOICE_KINDS[kind] + if arg not in {c.name for c in getattr(voice, offered)}: + raise ValueError(f"{arg!r} is not a model Flash offers") + if name == "voice-model-remove": + if arg == getattr(voice, setting)(): + raise ValueError("That one is in use. Pick another first.") + getattr(voice, remove)(arg) + return voice_models(session) + if getattr(voice, installed)(arg): + ai.set_config_var(VOICE_SETTINGS[kind], arg) + return voice_models(session) + session.fetch_voice_model(kind, arg) + return voice_models(session) + if name == "lan": if session.switch_lan is None: raise ValueError("This Flash cannot reopen its server.") @@ -2250,9 +2414,10 @@ def do_POST(self) -> None: return length = int(self.headers.get("Content-Length") or 0) - limit = ( - MAX_UPLOAD_BODY if url.path == "/api/upload" else MAX_BODY_BYTES - ) + limit = { + "/api/upload": MAX_UPLOAD_BODY, + "/api/voice/hear": MAX_VOICE_BODY, + }.get(url.path, MAX_BODY_BYTES) if length > limit: self._json( {"error": "too large"}, HTTPStatus.REQUEST_ENTITY_TOO_LARGE @@ -2281,6 +2446,7 @@ def _post(self, path: str, body: dict) -> dict: chat, str(body.get("text", "")), mode=str(body.get("mode") or ""), files=[str(f) for f in body.get("files") or []], + heard=bool(body.get("voice")), ) return {"chat": chat.id, **sent} @@ -2293,6 +2459,31 @@ def _post(self, path: str, body: dict) -> dict: raise ValueError("that upload did not arrive whole") from None return workspace.keep_upload(str(body.get("name") or ""), data) + if path == "/api/voice/hear": + try: + pcm = base64.b64decode( + str(body.get("data") or ""), validate=True + ) + except (ValueError, TypeError): + raise ValueError( + "that recording did not arrive whole" + ) from None + text, why = voice.transcribe(pcm) + if why: + raise ValueError(why) + return { + "text": text, + **voice.voice_command(text, str(body.get("speaking") or "")), + } + + if path == "/api/voice/say": + wav, why = voice.synthesize( + voice.speakable(str(body.get("text") or "")) + ) + if why: + raise ValueError(why) + return {"audio": base64.b64encode(wav).decode("ascii")} + if path == "/api/answer": return { "ok": session.answer( diff --git a/flash/web/index.html b/flash/web/index.html index 4dd5b27..a7250b5 100644 --- a/flash/web/index.html +++ b/flash/web/index.html @@ -442,6 +442,49 @@ } #jump svg { width: 16px; height: 16px; } +/* Voice mode: the box gives way to an orb that listens and speaks. */ +body.voice-on .shell, body.voice-on #suggestions, body.voice-on #pending { display: none !important; } +#voice { display: flex; align-items: center; gap: 16px; padding: 14px 14px 14px 16px; border: 1px solid var(--border-strong); border-radius: 22px; background: var(--surface); box-shadow: var(--composer-shadow); animation: pop 0.2s var(--ease); } +#voice[hidden] { display: none; } +.voice-orb { --lvl: 0; position: relative; flex: none; width: 58px; height: 58px; border-radius: 50%; cursor: pointer; + background: radial-gradient(circle at 34% 30%, #ffe1bd 0%, #f0a36f 30%, var(--accent) 62%, color-mix(in srgb, var(--accent) 55%, #000) 100%); + box-shadow: 0 0 calc(8px + var(--lvl) * 44px) color-mix(in srgb, var(--accent) 55%, transparent); + transform: scale(calc(1 + var(--lvl) * 0.22)); transition: transform 0.08s linear, box-shadow 0.08s linear, filter 0.3s; } +.voice-orb span { position: absolute; inset: -7px; border-radius: 50%; border: 2px solid color-mix(in srgb, var(--accent) 45%, transparent); opacity: 0; } +#voice[data-state="thinking"] .voice-orb, #voice[data-state="hearing"] .voice-orb, #voice[data-state="setup"] .voice-orb { animation: orb-breathe 1.5s ease-in-out infinite; } +#voice[data-state="listening"] .voice-orb span, #voice[data-state="asking"] .voice-orb span { animation: orb-ring 1.8s ease-out infinite; } +#voice[data-state="paused"] .voice-orb { filter: saturate(0.25) brightness(0.8); } +@keyframes orb-breathe { 50% { transform: scale(0.9); filter: brightness(0.85); } } +@keyframes orb-ring { from { opacity: 0.8; transform: scale(0.9); } to { opacity: 0; transform: scale(1.35); } } +.voice-say { flex: 1; min-width: 0; display: flex; flex-direction: column; gap: 2px; } +.voice-say b { font-size: 15.5px; font-weight: 600; color: var(--text); } +.voice-say span { font-size: 13px; color: var(--text-3); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.voice-end { flex: none; } +@media (prefers-reduced-motion: reduce) { .voice-orb { transform: none; } } + +/* Settings, voice models. */ +.voice-models { max-width: 620px; } +.vm-missing { font: 12.5px var(--mono); color: var(--warn); margin: 0 0 12px; } +.vm-head { display: flex; align-items: baseline; gap: 10px; margin: 16px 0 8px; } +.vm-head:first-child, .vm-missing + .vm-head { margin-top: 0; } +.vm-head b { font-size: 14px; font-weight: 600; color: var(--text); } +.vm-head span { font-size: 13px; color: var(--text-3); } +.vm-list { border: 1px solid var(--border); border-radius: 14px; background: var(--surface); overflow: hidden; } +.vm-row { display: flex; align-items: center; gap: 12px; padding: 11px 12px 11px 14px; } +.vm-row + .vm-row { border-top: 1px solid var(--border); } +.vm-mark { width: 16px; flex: none; color: var(--accent); display: grid; place-items: center; } +.vm-mark svg { width: 16px; height: 16px; } +.vm-what { flex: 1; min-width: 0; display: flex; flex-direction: column; gap: 2px; } +.vm-what b { font-size: 14px; font-weight: 500; color: var(--text); } +.vm-dot { color: var(--text-3); margin: 0 3px; font-size: 11px; vertical-align: 1px; } +.vm-what small { font: 12px var(--mono); color: var(--text-3); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.vm-act { flex: none; display: flex; align-items: center; gap: 6px; } +.vm-act .btn.small { padding: 5px 12px; font-size: 13px; } +.vm-act .btn:disabled { opacity: 0.5; cursor: default; } +.vm-tag { font-size: 12px; font-weight: 500; color: var(--accent); background: color-mix(in srgb, var(--accent) 12%, transparent); padding: 3px 9px; border-radius: 999px; } +.vm-progress { font: 12.5px var(--mono); color: var(--text-2); } +.vm-row.current .vm-what b { color: var(--text); } + /* Composer ------------------------------------------------------------- */ #dock { flex: none; padding: 0 0 max(18px, env(safe-area-inset-bottom)); position: relative; } #composer { @@ -1058,6 +1101,8 @@ /* The message a search led to glows, and its words are marked. */ ::highlight(found) { background-color: color-mix(in srgb, var(--accent) 38%, transparent); color: var(--text); } +/* What voice mode is reading aloud right now. */ +::highlight(speaking) { background-color: color-mix(in srgb, var(--accent) 26%, transparent); color: var(--text); } .block.found { animation: found 2.2s var(--ease); border-radius: 14px; } @keyframes found { 0%, 30% { box-shadow: 0 0 0 8px color-mix(in srgb, var(--accent) 16%, transparent); background: color-mix(in srgb, var(--accent) 7%, transparent); } @@ -1223,6 +1268,11 @@

Good evening

+
@@ -1321,6 +1371,7 @@ camera: '', clip: '', slash: '', + voice: '', image: '', server: '', more: '', @@ -1444,6 +1495,15 @@ let projects = []; // { id, name, path, instructions, created } let hosts = []; // { name, url, up } let attached = []; // what goes with the next message: { name, size, image, url, id, uploading, node } +// Recording needs a secure page: this computer, or HTTPS. +const canRecord = !!(navigator.mediaDevices && navigator.mediaDevices.getUserMedia && window.isSecureContext); +const VOICE_RATE = 16000; // what Vosk listens at +const VOICE_LOUD = 500 / 32768; // the terminal's threshold for speech +const VOICE_PAUSE = 1.2; // seconds of quiet that end a turn +const VOICE_PATIENCE = 8; // seconds with no speech before pausing +const VOICE_LONGEST = 120; // seconds in one turn, at most + +const voice = { on: false, state: "", ctx: null, stream: null, level: 0, turn: null, tap: null, source: null, cut: false, setup: null, analyser: null }; let view = "chat"; // "chat" | "projects" | "project" | "settings" | "skills" | "extensions" let viewProject = null; // the project page shown, or the project a new chat starts in @@ -2334,6 +2394,9 @@ }; body.appendChild(row); + body.insertAdjacentHTML("beforeend", `

Voice models

What voice mode listens and speaks with. Both run on this computer, so pick lighter ones for a weaker machine.

`); + drawVoiceModels(body.querySelector(".voice-models")); + body.insertAdjacentHTML("beforeend", `

Network

Who can reach Flash. A browser still has to open the link, with its random token, to get in.

`); renderLan(body.querySelector(".lan-box")); @@ -2621,6 +2684,62 @@

Install an extension

if (!r.scenes.length) grid.insertAdjacentHTML("afterend", `

No scenes found.

`); } +// Voice models: the listening and speaking models voice mode can use, +// each with its size and who it suits, to download, switch to, or free +// the space of. +const fmtBytes = (n) => (n >= 1e9 ? `${(n / 1e9).toFixed(1)} GB` : `${Math.round(n / 1e6)} MB`); +const VOICE_GROUPS = [["listening", "Listening", "Turns what you say into text."], ["speaking", "Speaking", "Reads Flash's replies aloud."]]; + +async function drawVoiceModels(box = document.querySelector(".voice-models")) { + if (!box) return; + let r; + try { r = await command("voice-models"); } catch (error) { showFailure(error); return; } + if (!box.isConnected) return; + box.textContent = ""; + if (r.missing.length) box.appendChild(el("p", "vm-missing")).textContent = r.install; + for (const [kind, title, sub] of VOICE_GROUPS) { + box.insertAdjacentHTML("beforeend", `
${title}${sub}
`); + const list = box.appendChild(el("div", "vm-list")); + for (const m of r[kind]) { + const fetching = r.downloading && r.downloading.kind === kind && r.downloading.name === m.name; + const row = list.appendChild(el("div", "vm-row" + (m.current ? " current" : ""), `${m.current ? icon("check") : ""}`)); + row.dataset.key = `${kind}:${m.name}`; + row.querySelector("b").innerHTML = `${fmtBytes(m.size)} • ${esc(m.label)}`; + row.querySelector("small").textContent = m.name; + const act = row.querySelector(".vm-act"); + if (fetching) { act.innerHTML = `Downloading`; continue; } + if (m.current) { act.innerHTML = `In use`; continue; } + const pick = el("button", "btn small", m.installed ? "Use" : "Download"); + pick.type = "button"; + pick.disabled = !!r.downloading || (!m.installed && r.missing.length > 0); + pick.onclick = async () => { + pick.disabled = true; + try { await command("voice-model", m.name, null, { kind }); drawVoiceModels(box); } + catch (error) { pick.disabled = false; showFailure(error); } + }; + act.appendChild(pick); + if (m.installed) { + const drop = button("icon-btn", "x", `Remove, freeing ${fmtBytes(m.size)}`, "", () => confirmed(drop, async () => { + try { await command("voice-model-remove", m.name, null, { kind }); drawVoiceModels(box); } catch (error) { showFailure(error); } + })); + act.appendChild(drop); + } + } + } +} + +function voiceModelEvent(event) { + const box = document.querySelector(".voice-models"); + if (event.done) { + if (event.error) toast(`Could not download ${event.name}: ${event.error}`); + else toast(`${event.name} is ready and in use`); + if (box) drawVoiceModels(box); + return; + } + const slot = box && box.querySelector(`.vm-row[data-key="${CSS.escape(`${event.kind}:${event.name}`)}"] .vm-act`); + if (slot) slot.innerHTML = `${esc(event.label)} ${event.percent}%`; +} + // LAN ---------------------------------------------------------------------- // Whether this page came in over the network rather than from this @@ -3525,6 +3644,426 @@

Add a host

let follow = true; const nearBottom = () => logEl.scrollHeight - logEl.scrollTop - logEl.clientHeight < 100; +// --------------------------------------------------------------------------- +// Voice mode. The browser records; Flash's own models, on the computer +// it runs on, turn speech into text and replies back into speech, the +// same Vosk and Piper the terminal's voice mode uses. So a round is: +// listen until you pause, send what was said, wait for the reply, read +// it aloud, and listen again. Tap the orb to cut a reply short. + + +function setVoice(state, hint) { + voice.state = state; + const panel = $("#voice"); + panel.dataset.state = state; + const words = { + setup: ["Getting voice ready", "Downloading the voice models, once."], + listening: ["Listening", "Speak, then pause. Say “stop voice” to finish."], + hearing: ["Catching that", ""], + thinking: ["Flash is working", pendingAsk() ? "It is asking you something on screen." : "It will answer out loud."], + speaking: ["Speaking", "Say \u201cinterrupt\u201d, or tap the orb, to cut in."], + paused: ["Paused", "Say \u201cresume\u201d, or tap the orb."], + asking: ["Flash is asking", "Say \u201cyes\u201d or \u201cno\u201d."], + }[state] || [state, ""]; + panel.querySelector(".voice-state").textContent = words[0]; + panel.querySelector(".voice-hint").textContent = hint ?? words[1]; +} + +async function startVoice() { + if (voice.on) return; + if (!canRecord) { toast("Voice mode needs this page opened on this computer, or over HTTPS"); return; } + let status; + try { status = await command("voice-status"); } catch (error) { showFailure(error); return; } + if (status.missing.length) { toast(status.install); return; } + voice.on = true; + voice.cut = false; + document.body.classList.add("voice-on"); + $("#voice").hidden = false; + if (!status.ready) { + setVoice("setup"); + const why = await new Promise((resolve) => { + voice.setup = resolve; + command("voice-setup").then((r) => { if (r.ready) resolve(""); }).catch((error) => resolve(error.message)); + }); + voice.setup = null; + if (why) { toast(`Voice mode could not get ready: ${why}`); endVoice(); return; } + if (!voice.on) return; + } + try { + voice.stream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } }); + } catch (_) { + toast("Flash could not use the microphone. Allow it for this page, then try again."); + endVoice(); + return; + } + if (!voice.on) { voice.stream.getTracks().forEach((t) => t.stop()); return; } + voice.ctx = new AudioContext(); + const source = voice.ctx.createMediaStreamSource(voice.stream); + // A script processor, not a worklet: it needs no module file, and a + // chunk of a few thousand samples at a time is plenty for speech. + const tap = voice.ctx.createScriptProcessor(4096, 1, 1); + const hush = voice.ctx.createGain(); + hush.gain.value = 0; + source.connect(tap); + tap.connect(hush); + hush.connect(voice.ctx.destination); + tap.onaudioprocess = (e) => hearChunk(e.inputBuffer.getChannelData(0)); + voice.analyser = voice.ctx.createAnalyser(); + voice.analyser.fftSize = 512; + voice.analyser.connect(voice.ctx.destination); + drawOrb(); + voiceLoop(); +} + +function endVoice() { + if (!voice.on) return; + voice.on = false; + if (voice.source) { try { voice.source.stop(); } catch (_) { /* already over */ } } + if (voice.turn) voice.turn.finish(null); + if (voice.tap) voice.tap(); + if (voice.setup) voice.setup(""); + if (voice.stream) voice.stream.getTracks().forEach((t) => t.stop()); + if (voice.ctx) voice.ctx.close().catch(() => {}); + Object.assign(voice, { ctx: null, stream: null, turn: null, tap: null, source: null, analyser: null, state: "" }); + $("#voice").hidden = true; + document.body.classList.remove("voice-on"); + updateSend(); + input.focus(); +} + +function voiceSetupEvent(event) { + if (!voice.on || voice.state !== "setup") return; + if (event.done) { if (voice.setup) voice.setup(event.error || ""); return; } + setVoice("setup", `Downloading the ${event.label}: ${event.percent}%`); +} + +// One turn of listening: resolves to the audio, 16 kHz 16-bit, once the +// speaker pauses, or to null if nobody spoke in time. OPTS tune it: how +// long a pause ends the turn, how long to wait for speech at all, the +// longest turn, and how loud speech has to be. +function listenOnce(opts = {}) { + const o = { pause: VOICE_PAUSE, patience: VOICE_PATIENCE, longest: VOICE_LONGEST, loud: VOICE_LOUD, ...opts }; + if (voice.turn) voice.turn.finish(null); + return new Promise((resolve) => { + const turn = { o, parts: [], heard: false, quiet: 0, elapsed: 0, finish: (pcm) => { if (voice.turn === turn) voice.turn = null; resolve(pcm); } }; + voice.turn = turn; + }); +} + +function hearChunk(samples) { + let sum = 0; + for (let i = 0; i < samples.length; i++) sum += samples[i] * samples[i]; + const rms = Math.sqrt(sum / samples.length); + if (voice.state === "listening" || voice.state === "asking") voice.level = Math.min(1, rms * 12); + const turn = voice.turn; + if (!turn) return; + // Down to Vosk's rate: each output sample the average of the input + // samples it stands for. + const ratio = voice.ctx.sampleRate / VOICE_RATE; + const out = new Int16Array(Math.floor(samples.length / ratio)); + for (let i = 0; i < out.length; i++) { + const from = Math.floor(i * ratio), to = Math.min(samples.length, Math.floor((i + 1) * ratio)); + let total = 0; + for (let j = from; j < to; j++) total += samples[j]; + const value = total / Math.max(1, to - from); + out[i] = Math.max(-32768, Math.min(32767, Math.round(value * 32767))); + } + const step = samples.length / voice.ctx.sampleRate; + turn.elapsed += step; + const loud = rms >= turn.o.loud; + // Waiting for speech, only the last second or so is kept, so a long + // wait does not become a long recording. + if (!turn.heard && !loud) { + turn.parts.push(out); + if (turn.parts.length > 6) turn.parts.shift(); + } else turn.parts.push(out); + if (loud) { turn.heard = true; turn.quiet = 0; } else turn.quiet += step; + const spoken = turn.heard ? turn.parts.length * out.length / VOICE_RATE : 0; + if (turn.heard && (turn.quiet >= turn.o.pause || spoken >= turn.o.longest)) { + const pcm = new Int16Array(turn.parts.reduce((n, part) => n + part.length, 0)); + let at = 0; + for (const part of turn.parts) { pcm.set(part, at); at += part.length; } + turn.finish(pcm); + } else if (!turn.heard && turn.elapsed >= turn.o.patience) turn.finish(null); +} + +function base64Of(bytes) { + let text = ""; + for (let i = 0; i < bytes.length; i += 0x8000) text += String.fromCharCode.apply(null, bytes.subarray(i, i + 0x8000)); + return btoa(text); +} + +// What was said, and what it asks of voice mode. SPEAKING is the line +// Flash is reading out, so its own words coming back through the +// microphone are not taken for an interruption. Null if it failed. +async function hear(pcm, speaking = "") { + try { return await api("/api/voice/hear", { data: base64Of(new Uint8Array(pcm.buffer)), speaking }); } + catch (error) { if (voice.on) toast(error.message); return null; } +} + +// Paused: waits for "resume", or a tap on the orb, and listens for +// nothing else, so a conversation in the room is not sent to Flash. +// Resolves false if voice mode ended meanwhile. +async function pausedUntilResume() { + setVoice("paused"); + for (;;) { + const tapped = new Promise((resolve) => { voice.tap = () => { voice.tap = null; resolve("tap"); }; }); + const which = await Promise.race([tapped, listenOnce({ pause: 0.8, patience: Infinity, longest: 6 })]); + if (!voice.on) return false; + if (which === "tap") { if (voice.turn) voice.turn.finish(null); return true; } + voice.tap = null; + if (!which) continue; + const heard = await hear(which); + if (!voice.on) return false; + if (heard && heard.exit) { endVoice(); toast("Voice mode off"); return false; } + if (heard && heard.resume) return true; + } +} + +async function voiceLoop() { + while (voice.on) { + setVoice("listening"); + const pcm = await listenOnce(); + if (!voice.on) return; + if (!pcm) { if (!(await pausedUntilResume())) return; continue; } + setVoice("hearing"); + const heard = await hear(pcm); + if (!voice.on) return; + if (!heard || !heard.text || heard.alone) continue; + if (heard.exit) { endVoice(); toast("Voice mode off"); return; } + if (heard.pause) { if (!(await pausedUntilResume())) return; continue; } + setVoice("thinking"); + const reply = await sendAndWait(heard.text); + if (!voice.on) return; + if (!reply) continue; + const how = await readAloud(reply); + if (how === "exit") { endVoice(); toast("Voice mode off"); return; } + if (how === "pause" && !(await pausedUntilResume())) return; + } +} + +// Sends TEXT as any message is sent, and resolves to the reply on the +// page once the turn it starts is over, or null if there was none. A +// permission Flash asks for on the way is asked out loud, and answered +// by voice. +async function sendAndWait(text) { + const before = current; + const startedAt = current && chats.get(current) ? chats.get(current).log.length : 0; + await send(text); + const asked = new Set(); + for (;;) { + await new Promise((r) => setTimeout(r, 250)); + if (!voice.on) return null; + const card = pendingAsk(); + if (card && !asked.has(card)) { + asked.add(card); + await askAloud(card); + if (!voice.on) return null; + setVoice("thinking"); + continue; + } + const chat = chats.get(current); + if (!chat) continue; + const from = current === before ? startedAt : 0; + if (chat.busy || chat.queued || chat.log.length <= from) continue; + const replies = chat.log.slice(from).filter((e) => e.type === "assistant" && e.text && e.text.trim()); + if (!replies.length) return null; + const said = replies[replies.length - 1].text; + const shown = [...messagesEl.querySelectorAll("[data-reply]")].reverse().find((n) => n.dataset.raw === said); + return shown || null; + } +} + +// Reads a permission question out and waits for a spoken yes or no. After +// a few tries without one, it is left on screen to answer by hand. +const YES = /^(yes|yeah|yep|yup|sure|ok|okay|allow|allowed|go ahead|do it|approve)\b/; +const NO = /^(no|nope|nah|deny|denied|don't|do not|stop|cancel|reject)\b/; +async function askAloud(card) { + setVoice("asking"); + const question = (card.querySelector(".q") || card).textContent.trim(); + // What it wants to run or change, so it can be judged by ear; a long + // one is cut, the rest is on screen. + const what = (card.querySelector("pre") || {}).textContent || ""; + const detail = what.trim().replace(/\s+/g, " "); + const said = detail.length > 160 ? `${detail.slice(0, 160)}, and more on screen` : detail; + await sayLine(`Flash asks: ${question} ${said ? `${said}.` : ""} Say yes or no.`); + for (let tries = 0; tries < 3 && voice.on && card.isConnected && !card.classList.contains("done"); tries++) { + setVoice("asking"); + const pcm = await listenOnce(); + if (!voice.on || !card.isConnected || card.classList.contains("done")) return; + if (!pcm) continue; + const heard = await hear(pcm); + if (!voice.on) return; + if (!heard) continue; + if (heard.exit) { endVoice(); toast("Voice mode off"); return; } + const said = heard.text.toLowerCase().trim(); + if (YES.test(said)) { answerPending("y"); return; } + if (NO.test(said)) { answerPending("n"); return; } + await sayLine("Say yes or no."); + } +} + +// A reply's sentences as they read on screen, each with the range of the +// page it covers, so the one being said can be marked where it sits. +// Code blocks and tables are shown, not read. +function sentencesOf(root) { + const skip = "pre, table, .katex, math, svg, .actions"; + const spans = []; + let text = "", home = null; + const walk = document.createTreeWalker(root, NodeFilter.SHOW_TEXT, { + acceptNode: (n) => (n.parentElement.closest(skip) ? NodeFilter.FILTER_REJECT : NodeFilter.FILTER_ACCEPT), + }); + for (let n = walk.nextNode(); n; n = walk.nextNode()) { + const block = n.parentElement.closest("p, li, h1, h2, h3, h4, h5, h6, blockquote, dt, dd") || root; + if (home && block !== home) text += "\n"; + home = block; + spans.push({ node: n, from: text.length }); + text += n.nodeValue; + } + const point = (offset, closing) => { + for (let i = spans.length - 1; i >= 0; i--) { + const s = spans[i]; + if (closing ? offset > s.from : offset >= s.from) return [s.node, Math.min(offset - s.from, s.node.nodeValue.length)]; + } + return [spans[0].node, 0]; + }; + const out = []; + for (const m of text.matchAll(/[^\n]+?(?:[.!?]+["')\]]*(?=\s|$)|$)/gm)) { + const lead = m[0].length - m[0].trimStart().length; + const body = m[0].trim(); + // A sentence of emoji, or of punctuation, has nothing to say. + if (!/[\p{L}\p{N}]/u.test(body)) continue; + const a = m.index + lead, b = a + body.length; + const range = document.createRange(); + range.setStart(...point(a, false)); + range.setEnd(...point(b, true)); + out.push({ text: body.replace(/\s+/g, " "), range }); + } + return out; +} + +// Marks what is being said, and keeps it in view as the reading moves on. +function markSpoken(range) { + if (!range || !window.CSS || !CSS.highlights) return; + CSS.highlights.set("speaking", new Highlight(range)); + const r = range.getBoundingClientRect(); + const box = logEl.getBoundingClientRect(); + if (r.top < box.top + 64 || r.bottom > box.bottom - 24) { + follow = false; + logEl.scrollBy({ top: r.top - box.top - box.height * 0.3, behavior: calm.matches ? "auto" : "smooth" }); + } +} +const unmarkSpoken = () => { if (window.CSS && CSS.highlights) CSS.highlights.delete("speaking"); }; + +// A line of speech, made on this computer, ready to play. +function makeSpeech(line) { + return api("/api/voice/say", { text: line }).then((r) => { + const raw = atob(r.audio); + const bytes = new Uint8Array(raw.length); + for (let i = 0; i < raw.length; i++) bytes[i] = raw.charCodeAt(i); + return voice.ctx.decodeAudioData(bytes.buffer); + }); +} + +function playSpeech(audio) { + return new Promise((resolve) => { + const source = voice.ctx.createBufferSource(); + source.buffer = audio; + source.connect(voice.analyser); + source.onended = () => { if (voice.source === source) voice.source = null; resolve(); }; + voice.source = source; + source.start(); + }); +} + +async function sayLine(line) { + try { const audio = await makeSpeech(line); if (voice.on) await playSpeech(audio); } catch (error) { if (voice.on) toast(error.message); } +} + +// Stops the reading for REASON: "tap", "interrupt", "pause" or "exit". +function cutReading(reason) { + voice.cut = reason; + if (voice.source) { try { voice.source.stop(); } catch (_) { /* already over */ } } +} + +// While a reply is read out, keeps an ear open for the interrupt word, +// "pause", or "stop voice". Only a louder sound counts as speech here, so +// the reply itself, softened by the browser's echo cancelling, does not. +function listenForInterruptions(line) { + let stopped = false; + (async () => { + while (!stopped && voice.on && voice.state === "speaking") { + const pcm = await listenOnce({ pause: 0.6, patience: Infinity, longest: 5, loud: VOICE_LOUD * 2.5 }); + if (stopped || !pcm || !voice.on) return; + const heard = await hear(pcm, line()); + if (stopped || !heard) continue; + if (heard.exit) { cutReading("exit"); return; } + if (heard.pause) { cutReading("pause"); return; } + if (heard.interrupt) { cutReading("interrupt"); return; } + } + })(); + return () => { stopped = true; if (voice.turn) voice.turn.finish(null); }; +} + +// Reads a whole reply aloud, a sentence at a time: the next one is made +// while this one plays, so the gaps stay short. Resolves to how it ended: +// "done", or what cut it short. +async function readAloud(node) { + const parts = sentencesOf(node.querySelector(".assistant") || node); + if (!parts.length) parts.push({ text: "That reply is code. It is on screen.", range: null }); + setVoice("speaking"); + voice.cut = false; + let now = ""; + const stopListening = listenForInterruptions(() => now); + let next = makeSpeech(parts[0].text); + try { + for (let i = 0; i < parts.length; i++) { + let audio; + try { audio = await next; } catch (error) { if (voice.on) toast(error.message); return "done"; } + if (i + 1 < parts.length) next = makeSpeech(parts[i + 1].text); + if (!voice.on) return "exit"; + if (voice.cut) return voice.cut; + now = parts[i].text; + markSpoken(parts[i].range); + await playSpeech(audio); + if (!voice.on) return "exit"; + if (voice.cut) return voice.cut; + } + return "done"; + } finally { + stopListening(); + next.catch(() => {}); + unmarkSpoken(); + } +} + +// The orb: what the microphone hears while listening, what Flash says +// while speaking. +function drawOrb() { + if (!voice.on) return; + const orb = $("#voice .voice-orb"); + let level = 0; + if (voice.state === "listening") level = voice.level; + else if (voice.state === "speaking" && voice.analyser) { + const data = new Uint8Array(voice.analyser.fftSize); + voice.analyser.getByteTimeDomainData(data); + let sum = 0; + for (const v of data) sum += ((v - 128) / 128) ** 2; + level = Math.min(1, Math.sqrt(sum / data.length) * 4); + } + voice.shown = (voice.shown || 0) * 0.7 + level * 0.3; + orb.style.setProperty("--lvl", voice.shown.toFixed(3)); + requestAnimationFrame(drawOrb); +} + +$("#voice .voice-orb").onclick = () => { + if (voice.state === "speaking") cutReading("tap"); + else if (voice.state === "paused" && voice.tap) voice.tap(); + else if (voice.state === "listening" && voice.turn && voice.turn.heard) voice.turn.quiet = VOICE_PAUSE; +}; +$("#voice .voice-end").innerHTML = icon("x"); +$("#voice .voice-end").onclick = endVoice; + // Two rings out from the send button, the second a beat behind. function shockwave(from) { if (calm.matches) return; @@ -3619,6 +4158,8 @@

Add a host

if (event.seq) lastSeq = event.seq; if (event.type === "chats" || event.type === "status" || event.type === "projects") { refreshLite(); return; } if (event.type === "update") { info.update = event.update; onUpdate(); return; } + if (event.type === "voice-setup") { voiceSetupEvent(event); return; } + if (event.type === "voice-model") { voiceModelEvent(event); return; } if (event.type === "browsers") { if (view === "settings" && settingsTab === "security") renderBrowsers(); return; } const chat = chats.get(event.chat); @@ -3983,7 +4524,8 @@

Add a host

const typed = !!input.value.trim() || attached.length > 0; // Working with nothing typed: Stop. Working with something typed: it // steers. Otherwise: send. - const state = busy ? (typed ? "steer" : "stop") : "up"; + // Idle with nothing typed, where a browser can record: voice mode. + const state = busy ? (typed ? "steer" : "stop") : (typed || !canRecord ? "up" : "voice"); const send = $("#send"); send.classList.toggle("stop", state === "stop"); document.body.classList.toggle("working", !!busy && view === "chat"); @@ -3993,11 +4535,11 @@

Add a host

send.dataset.state = state; send.innerHTML = icon(state); } - const tip = { stop: ["Stop", "Esc"], steer: ["Steer: Flash reads it at its next step. Ctrl Enter runs it after instead", "Enter"], up: ["Send", "Enter"] }[state]; + const tip = { stop: ["Stop", "Esc"], steer: ["Steer: Flash reads it at its next step. Ctrl Enter runs it after instead", "Enter"], up: ["Send", "Enter"], voice: ["Voice mode: talk with Flash", ""] }[state]; send.dataset.tip = tip[0]; send.dataset.keys = tip[1]; - send.setAttribute("aria-label", state === "stop" ? "Stop" : state === "steer" ? "Steer" : "Send"); - send.disabled = (!busy && !typed) || (typed && uploading()); + send.setAttribute("aria-label", { stop: "Stop", steer: "Steer", voice: "Voice mode" }[state] || "Send"); + send.disabled = (state === "up" && !typed) || (typed && uploading()); if (busy && view === "chat") input.placeholder = "Steer Flash, or Ctrl Enter to queue"; else if (input.placeholder.startsWith("Steer")) input.placeholder = "Reply to Flash"; } @@ -4101,7 +4643,8 @@

Add a host

// The chat has to exist here before anything is said in it, or its // first events would arrive for a chat this page has never seen. if (!current) await createChat(viewProject || ""); - await api("/api/send", { chat: current, text, files }).catch(showFailure); + // From voice mode, the reply will be heard: the model is told so. + await api("/api/send", { chat: current, text, files, voice: voice.on }).catch(showFailure); } async function createChat(project = "") { @@ -4875,6 +5418,7 @@

Add a host

{ label: "Background", keys: "", slash: "/background", run: () => showSettings("general") }, { label: "Extensions", keys: "", slash: "/extensions", run: () => showPage("extensions") }, { label: "Import memory from another AI", keys: "", slash: "/import", run: () => { showSettings("memory"); importMemory(); } }, + { label: "Voice mode", keys: "", slash: "/voice", run: () => startVoice() }, { label: "Signed-in browsers", keys: "", slash: "/security", run: () => showSettings("security") }, { label: "Search chats", keys: "Alt F", slash: "/search", run: () => openSearch() }, { label: "Show or hide the file panel", keys: "Alt V", slash: "/file", run: toggleViewer }, @@ -5265,7 +5809,11 @@

Add a host

} }); -$("#composer").addEventListener("submit", (e) => { e.preventDefault(); if (input.value.trim() || attached.length) send(input.value); else stopReply(); }); +$("#composer").addEventListener("submit", (e) => { + e.preventDefault(); + if (input.value.trim() || attached.length) send(input.value); + else if (!stopReply() && canRecord) startVoice(); +}); $("#composer").addEventListener("mousedown", (e) => { if (e.target === e.currentTarget) { e.preventDefault(); input.focus(); } }); $("#new-chat").onclick = newChat; $("#top-new").onclick = newChat; @@ -5304,6 +5852,7 @@

Add a host

// An overlay on its way out counts as closed, so a key pressed during // its exit (Ctrl K straight after Esc) is not swallowed. const up = (node) => !node.hidden && !node.classList.contains("closing"); + if (voice.on && e.key === "Escape") { e.preventDefault(); endVoice(); return; } if (up($("#help"))) { if (e.key === "Escape" || e.key === "?" || (mod && e.key === "/")) { e.preventDefault(); closeHelp(); } return; } if (up($("#phone"))) { if (e.key === "Escape") { e.preventDefault(); closePhone(); } return; } if (up($("#form"))) return; diff --git a/tests/conftest.py b/tests/conftest.py index 8d62e0f..13448be 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,8 +1,11 @@ # pylint: disable=C0114,C0116 +import os + import pytest from flash import ( + ai, extensions, learning, memory, @@ -10,6 +13,7 @@ skills, terminal, tools, + voice, workspace, ) @@ -50,6 +54,14 @@ def isolated_home(tmp_path_factory, monkeypatch): monkeypatch.setattr(skills, "FLASH_DIR", home / ".flash") # The web UI's saved hosts, projects, and chats. monkeypatch.setattr(workspace, "FLASH_DIR", home / ".flash") + # Voice settings and models too: the developer's own picks, loaded + # from ~/.flash.env when ai was imported, would otherwise decide + # which model a test expects. And a setting a test saves lands in + # the temp home, never in the developer's real env file. + for name in [n for n in os.environ if n.startswith("VOICE_")]: + monkeypatch.delenv(name) + monkeypatch.setattr(voice, "MODELS_DIR", home / ".flash" / "models") + monkeypatch.setattr(ai, "ENV_PATH", str(home / ".flash.env")) learning.refresh() learning.reset() yield home diff --git a/tests/test_voice.py b/tests/test_voice.py index 8968878..dd3fd2f 100644 --- a/tests/test_voice.py +++ b/tests/test_voice.py @@ -522,3 +522,197 @@ def test_the_speaker_is_reloaded_when_the_voice_changes( assert _load_speaker()[1] == 22050 # nosec B101 assert loaded[1] == 16000 # nosec B101 + + +# --- For the web UI: audio in from the browser, speech out as WAV ---------- + + +def test_transcribe_reads_what_the_browser_recorded(monkeypatch): + _fake_audio(monkeypatch) + recorded = _pcm(3000) * 3 + b"\x01" # a stray odd byte is dropped + + text, why = voice.transcribe(recorded) + + assert (text, why) == ("run the tests", "") # nosec B101 + + +def test_transcribe_explains_a_missing_package(monkeypatch): + monkeypatch.setitem(sys.modules, "vosk", None) + + text, why = voice.transcribe(_pcm(3000)) + + assert text == "" # nosec B101 + assert "vosk" in why # nosec B101 + + +@pytest.mark.parametrize("streaming", [True, False]) +def test_synthesize_hands_back_a_wav(monkeypatch, tmp_path, streaming): + import io + import wave + + _fake_piper(monkeypatch, tmp_path, streaming=streaming) + monkeypatch.setattr(voice, "_speaker", None) + + audio, why = voice.synthesize("all done") + + assert why == "" # nosec B101 + with wave.open(io.BytesIO(audio)) as wav: + assert wav.getframerate() == 16000 # nosec B101 + assert wav.getnchannels() == 1 # nosec B101 + assert wav.readframes(8) == _pcm(1000, 8) # nosec B101 + + +def test_synthesize_says_nothing_about_nothing(): + assert voice.synthesize(" ") == (b"", "") # nosec B101 + + +def test_synthesize_explains_a_missing_package(monkeypatch, tmp_path): + monkeypatch.setattr(voice, "MODELS_DIR", tmp_path) + onnx = tmp_path / f"{voice.piper_voice()}.onnx" + onnx.write_bytes(b"x") + onnx.with_suffix(".onnx.json").write_text("{}", encoding="utf-8") + monkeypatch.setattr(voice, "_speaker", None) + monkeypatch.setitem(sys.modules, "piper", None) + + audio, why = voice.synthesize("hello") + + assert audio == b"" # nosec B101 + assert "piper" in why # nosec B101 + + +def test_a_line_is_made_fit_to_hear(): + assert voice.speakable( # nosec B101 + "See https://example.com/docs\n for more." + ) == "See a link for more." + + +@pytest.mark.parametrize("catalogue", [voice.LISTENING, voice.SPEAKING]) +def test_settings_offers_the_defaults_lightest_first(catalogue): + names = [c.name for c in catalogue] + sizes = [c.size for c in catalogue] + + assert sizes == sorted(sizes) # nosec B101 + assert voice.DEFAULT_VOSK_MODEL in names or ( # nosec B101 + voice.DEFAULT_PIPER_VOICE in names + ) + assert all(c.label for c in catalogue) # nosec B101 + + +def test_one_model_downloads_by_name(tmp_path, monkeypatch): + monkeypatch.setattr(voice, "MODELS_DIR", tmp_path) + fetched = [] + + def fetch(url, out, label, on_progress): + fetched.append(url) + if url.endswith(".zip"): + with zipfile.ZipFile(out, "w") as bundle: + bundle.writestr("vosk-model-en-us-0.22-lgraph/am/x", "x") + else: + out.write_text("{}") + on_progress(label, 100) + return "" + + monkeypatch.setattr(voice, "_download", fetch) + + assert voice.download_listening( # nosec B101 + "vosk-model-en-us-0.22-lgraph", lambda *_: None + ) == "" + assert voice.download_voice( # nosec B101 + "en_US-ryan-high", lambda *_: None + ) == "" + assert voice.listening_installed( # nosec B101 + "vosk-model-en-us-0.22-lgraph" + ) + assert voice.voice_installed("en_US-ryan-high") # nosec B101 + assert fetched[1].endswith( # nosec B101 + "/en/en_US/ryan/high/en_US-ryan-high.onnx" + ) + + voice.remove_listening("vosk-model-en-us-0.22-lgraph") + voice.remove_voice("en_US-ryan-high") + + assert not voice.listening_installed( # nosec B101 + "vosk-model-en-us-0.22-lgraph" + ) + assert not voice.voice_installed("en_US-ryan-high") # nosec B101 + + +def test_only_offered_models_can_be_removed(tmp_path, monkeypatch): + monkeypatch.setattr(voice, "MODELS_DIR", tmp_path) + + with pytest.raises(ValueError): + voice.remove_listening("../../somewhere") + with pytest.raises(ValueError): + voice.remove_voice("not-a-voice") + + +def test_the_web_needs_no_sounddevice(monkeypatch): + monkeypatch.setattr( + voice, "missing_packages", + lambda: ["vosk", "sounddevice"], + ) + + assert voice.web_missing() == ["vosk"] # nosec B101 + + +def test_emojis_are_left_unsaid(): + said = voice.speakable( + "Done \u2705 all good \U0001F389\U0001F44D\U0001F3FD!" + ) + shipped = for_speech("Shipped \U0001F680 today.") + + assert said == "Done all good !" # nosec B101 + assert "\U0001F680" not in shipped # nosec B101 + + +@pytest.mark.parametrize("said, asks", [ + ("interrupt", {"interrupt": True, "alone": True}), + ("stop talking please", {"interrupt": True, "alone": False}), + ("pause", {"pause": True}), + ("hold on", {"pause": True}), + ("uh resume", {"resume": True}), + ("stop voice", {"exit": True}), + ("resume the download", {"resume": True, "pause": False, "exit": False}), + ("fix the flaky test", {"interrupt": False, "pause": False, + "resume": False, "exit": False}), +]) +def test_what_speech_asks_of_voice_mode(said, asks): + got = voice.voice_command(said) + + assert {k: got[k] for k in asks} == asks # nosec B101 + + +def test_the_reply_saying_the_word_is_not_an_interruption(): + got = voice.voice_command("interrupt", speaking="Say interrupt to stop.") + + assert got["interrupt"] is False # nosec B101 + + +@pytest.mark.parametrize("written, said", [ + ("is_expired >= now", "is expired at least now"), + ("Nested [[1, 2]] list", "Nested 1, 2 list"), + ("`npm i` | grep x && go", "npm i grep x and go"), + ("C++ -> fast", "C plus plus to fast"), + ("x\u00b2 = \u00bd", "x2 equals 1 over 2"), + ("a != b \u2192 c", "a is not b to c"), + ("\u2022 one \u2022 two", "one two"), + ("plain words.", "plain words."), +]) +def test_text_is_made_sayable(written, said): + assert voice.fit_for_voice(written) == said # nosec B101 + + +def test_what_reaches_piper_is_sayable(monkeypatch, tmp_path): + import logging + + _fake_piper(monkeypatch, tmp_path, streaming=True) + monkeypatch.setattr(voice, "_speaker", None) + heard = [] + voice_obj, _rate = _load_speaker() + voice_obj.synthesize_stream_raw = lambda text: heard.append(text) or [] + + list(voice._pcm_chunks(voice_obj, "Nested [[ae]] is_ok")) + + assert heard == ["Nested ae is ok"] # nosec B101 + # Piper's own complaints about sounds it lacks stay out of the way. + assert logging.getLogger("piper").level == logging.ERROR # nosec B101 diff --git a/tests/test_web.py b/tests/test_web.py index 97caf57..4040980 100644 --- a/tests/test_web.py +++ b/tests/test_web.py @@ -2470,6 +2470,237 @@ def test_a_chat_address_without_the_token_is_the_expired_page( assert b" web.MAX_BODY_BYTES + + # Anywhere else, the ordinary limit still holds. Refused on the + # length it claims, before anything is read. + port = server.port + conn = http.client.HTTPConnection("127.0.0.1", port, timeout=5) + conn.putrequest("POST", "/api/voice/say") + for name, value in ( + ("Host", f"127.0.0.1:{port}"), ("X-Flash-Token", server.token), + ("Content-Type", "application/json"), + ("Content-Length", str(web.MAX_BODY_BYTES + 1)), + ): + conn.putheader(name, value) + conn.endheaders(b"{}") + response = conn.getresponse() + response.read() + conn.close() + assert response.status == 413 + + def test_a_reply_is_spoken_as_wav(self, server, monkeypatch): + monkeypatch.setattr(web.voice, "synthesize", + lambda text: (b"RIFF" + text.encode(), "")) + + status, body = request(server, "POST", "/api/voice/say", + {"text": "done"}) + + assert status == 200 + assert base64.b64decode(json.loads(body)["audio"]) == b"RIFFdone" + + def test_the_voice_needs_are_reported(self, monkeypatch): + monkeypatch.setattr(web.voice, "web_missing", lambda: ["vosk"]) + monkeypatch.setattr(web.voice, "models_present", lambda: False) + + said = web.command(web.Session(), {"name": "voice-status"}) + + assert said["missing"] == ["vosk"] and said["ready"] is False + assert "vosk" in said["install"] + + def test_a_spoken_line_says_links_as_links(self, server, monkeypatch): + said = [] + monkeypatch.setattr(web.voice, "synthesize", + lambda text: (said.append(text) or b"RIFF", "")) + + request(server, "POST", "/api/voice/say", + {"text": "Read https://x.dev/a now."}) + + assert said == ["Read a link now."] + + def test_settings_lists_the_models_and_where_each_stands( + self, monkeypatch, + ): + monkeypatch.setattr(web.voice, "listening_installed", + lambda name: name == web.voice.DEFAULT_VOSK_MODEL) + monkeypatch.setattr(web.voice, "voice_installed", lambda name: False) + monkeypatch.setattr(web.voice, "web_missing", lambda: []) + + listed = web.command(web.Session(), {"name": "voice-models"}) + + first = listed["listening"][0] + assert first["installed"] and first["current"] + assert first["size"] == 41_205_931 and "weaker" in first["label"] + assert not any(v["installed"] for v in listed["speaking"]) + assert listed["downloading"] is None + + def test_picking_an_installed_model_uses_it(self, monkeypatch): + saved = {} + monkeypatch.setattr(ai, "set_config_var", + lambda name, value: saved.update({name: value})) + monkeypatch.setattr(web.voice, "voice_installed", lambda name: True) + + web.command(web.Session(), { + "name": "voice-model", "arg": "en_US-ryan-high", + "kind": "speaking", + }) + + assert saved == {"VOICE_PIPER_VOICE": "en_US-ryan-high"} + + def test_picking_a_missing_model_downloads_then_uses_it( + self, monkeypatch, + ): + saved = {} + monkeypatch.setattr(ai, "set_config_var", + lambda name, value: saved.update({name: value})) + monkeypatch.setattr(web.voice, "listening_installed", lambda n: False) + + def fetch(name, progress): + progress("listening model", 40) + progress("listening model", 100) + return "" + + monkeypatch.setattr(web.voice, "download_listening", fetch) + session = web.Session() + drain = events_of(session) + + web.command(session, { + "name": "voice-model", "arg": "vosk-model-en-us-0.22", + "kind": "listening", + }) + wait_for(lambda: session.voice_job is None) + time.sleep(0.05) + + told = [e for e in drain() if e["type"] == "voice-model"] + assert [e.get("percent") for e in told[:2]] == [40, 100] + assert told[-1]["done"] is True and told[-1]["error"] == "" + assert saved == {"VOICE_VOSK_MODEL": "vosk-model-en-us-0.22"} + + def test_what_settings_refuses(self, monkeypatch): + session = web.Session() + in_use = web.voice.vosk_model() + + with pytest.raises(ValueError, match="in use"): + web.command(session, { + "name": "voice-model-remove", "arg": in_use, + "kind": "listening", + }) + with pytest.raises(ValueError, match="not a model Flash offers"): + web.command(session, { + "name": "voice-model", "arg": "../../x", + "kind": "listening", + }) + with pytest.raises(ValueError, match="listening or speaking"): + web.command(session, { + "name": "voice-model", "arg": in_use, "kind": "other", + }) + + def test_the_models_download_once_with_progress(self, monkeypatch): + gate = threading.Event() + + def fetch(progress): + progress("voice", 50) + progress("voice", 50) + progress("voice", 100) + gate.wait(2) + return "" + + monkeypatch.setattr(web.voice, "models_present", lambda: False) + monkeypatch.setattr(web.voice, "ensure_models", fetch) + session = web.Session() + drain = events_of(session) + + first = web.command(session, {"name": "voice-setup"}) + again = web.command(session, {"name": "voice-setup"}) + gate.set() + wait_for(lambda: not session.voice_setup) + time.sleep(0.05) + + assert first == again == {"ready": False} + told = [e for e in drain() if e["type"] == "voice-setup"] + assert [e.get("percent") for e in told[:2]] == [50, 100] + assert told[-1]["done"] is True and told[-1]["error"] == "" + assert len(told) == 3 + + def test_ready_models_need_no_download(self, monkeypatch): + monkeypatch.setattr(web.voice, "models_present", lambda: True) + + assert web.command(web.Session(), {"name": "voice-setup"}) == { + "ready": True, + } + + class TestAttachments: def upload(self, server, name, data): return request(server, "POST", "/api/upload", {