From 1cc9ff1365b3cd1af9af45761adcdf6930687442 Mon Sep 17 00:00:00 2001 From: "Nathan C." <149914029+Natuworkguy@users.noreply.github.com> Date: Sun, 27 Sep 2026 22:39:57 -0700 Subject: [PATCH 1/3] Add Voice for Web --- flash/ai.py | 12 +- flash/voice.py | 313 ++++++++++++++++++++---- flash/web.py | 201 +++++++++++++++- flash/web/index.html | 561 ++++++++++++++++++++++++++++++++++++++++++- tests/conftest.py | 12 + tests/test_voice.py | 194 +++++++++++++++ tests/test_web.py | 231 ++++++++++++++++++ 7 files changed, 1463 insertions(+), 61 deletions(-) diff --git a/flash/ai.py b/flash/ai.py index 6dd886e..5a3eb1f 100644 --- a/flash/ai.py +++ b/flash/ai.py @@ -157,12 +157,12 @@ === Voice Mode === The user is speaking to you, and your reply is read back to them out loud. -Keep it short and plain: whole sentences, no code blocks, tables, or long -lists unless they ask for one, because only the prose is spoken and the -rest is silently dropped. What they said reached you through speech -recognition, so expect missing punctuation and the occasional misheard -word; ask when a name, path, or command sounds wrong rather than acting on -a guess.""".rstrip() +Keep it short: two or three sentences unless they ask for more. Plain whole +sentences, no code blocks, tables, lists, or emojis unless they ask for +one, because only the prose is spoken and the rest is silently dropped. +What they said reached you through speech recognition, so expect missing +punctuation and the occasional misheard word; ask when a name, path, or +command sounds wrong rather than acting on a guess.""".rstrip() class FlashError(Exception): diff --git a/flash/voice.py b/flash/voice.py index d1ac8b3..7842ff5 100644 --- a/flash/voice.py +++ b/flash/voice.py @@ -7,14 +7,20 @@ """ import array +import io import json +import logging import math import os import re +import shutil import sys import threading +import unicodedata +import wave import zipfile from collections.abc import Callable +from dataclasses import dataclass from pathlib import Path from urllib.error import URLError from urllib.request import Request, urlopen @@ -133,6 +139,36 @@ def interrupt_word() -> str: MAX_TURN_SECONDS = 120.0 + +@dataclass(frozen=True) +class Choice: + """A model Settings offers: its name, download size, and who it suits.""" + + name: str + size: int + label: str + + +# What Settings offers to listen with, lightest first. Sizes are the +# downloads, measured from the servers that host them. +LISTENING = ( + Choice(DEFAULT_VOSK_MODEL, 41_205_931, + "Small and quick, for weaker machines"), + Choice("vosk-model-en-us-0.22-lgraph", 130_557_655, + "Balanced, noticeably more accurate"), + Choice("vosk-model-en-us-0.22", 1_913_365_522, + "Most accurate, for strong machines"), +) + +# And to speak with. +SPEAKING = ( + Choice("en_US-amy-low", 63_104_526, "Amy, fastest, for weaker machines"), + Choice(DEFAULT_PIPER_VOICE, 63_201_294, "Amy, balanced"), + Choice("en_US-lessac-high", 113_895_201, "Lessac, clearest"), + Choice("en_US-ryan-high", 120_786_792, + "Ryan, clearest, for strong machines"), +) + DOWNLOAD_TIMEOUT = 30 DOWNLOAD_CHUNK = 1 << 16 @@ -140,21 +176,30 @@ def interrupt_word() -> str: State = Callable[[str], None] -def vosk_model_dir() -> Path: - """Where the unpacked Vosk model lives once downloaded.""" +def vosk_model_dir(name: str = "") -> Path: + """Where an unpacked Vosk model lives once downloaded.""" - return MODELS_DIR / vosk_model() + return MODELS_DIR / (name or vosk_model()) -def piper_paths() -> tuple[Path, Path]: - """The Piper voice's network and its config file.""" +def piper_paths(name: str = "") -> tuple[Path, Path]: + """A Piper voice's network and its config file.""" - onnx = MODELS_DIR / f"{piper_voice()}.onnx" + onnx = MODELS_DIR / f"{name or piper_voice()}.onnx" return onnx, onnx.with_suffix(".onnx.json") -def _piper_urls() -> tuple[str, str]: +def listening_installed(name: str) -> bool: + return (vosk_model_dir(name) / "am").is_dir() + + +def voice_installed(name: str) -> bool: + onnx, config = piper_paths(name) + return onnx.is_file() and config.is_file() + + +def _piper_urls(name: str = "") -> tuple[str, str]: """The download addresses for the configured Piper voice. A voice is named locale-speaker-quality ("en_US-amy-medium"), and @@ -163,7 +208,7 @@ def _piper_urls() -> tuple[str, str]: name that is not shaped like a voice. """ - name = piper_voice() + name = name or piper_voice() locale, speaker, quality = name.split("-", 2) language = locale.split("_")[0].lower() base = f"{PIPER_BASE}/{language}/{locale}/{speaker}/{quality}/{name}.onnx" @@ -203,6 +248,16 @@ def missing_packages() -> list[str]: return missing +def web_missing() -> list[str]: + """What voice in the web UI still needs installed. + + Listening and speaking only: the browser has the microphone and the + speakers, so sounddevice is the terminal's business, not the page's. + """ + + return [p for p in missing_packages() if p != "sounddevice"] + + def _download(url: str, out: Path, label: str, on_progress: Progress) -> str: """Stream `url` to `out`, reporting percent complete as it goes.""" @@ -266,51 +321,72 @@ def ensure_models(on_progress: Progress) -> str: if models_present(): return "" + return (download_listening(vosk_model(), on_progress) + or download_voice(piper_voice(), on_progress)) + + +def download_listening(name: str, on_progress: Progress) -> str: + """Fetch and unpack the Vosk model NAME. Returns "" once it is in.""" + + if listening_installed(name): + return "" + MODELS_DIR.mkdir(parents=True, exist_ok=True) + archive = MODELS_DIR / f"{name}.zip" + why = _download( + f"{VOSK_BASE}/{name}.zip", archive, "listening model", on_progress, + ) + if why: + return why + + why = _unpack(archive, MODELS_DIR) + if why: + return why - if not (vosk_model_dir() / "am").is_dir(): - name = vosk_model() - archive = MODELS_DIR / f"{name}.zip" - why = _download( - f"{VOSK_BASE}/{name}.zip", - archive, - "listening model", - on_progress, + if not listening_installed(name): + return ( + f"{name} did not unpack into {vosk_model_dir(name)}. Check the " + "model name in VOICE_VOSK_MODEL." ) - if why: - return why + return "" - why = _unpack(archive, MODELS_DIR) - if why: - return why - if not (vosk_model_dir() / "am").is_dir(): - return ( - f"{name} did not unpack into " - f"{vosk_model_dir()}. Check the model name in " - "VOICE_VOSK_MODEL." - ) +def download_voice(name: str, on_progress: Progress) -> str: + """Fetch the Piper voice NAME, network and settings. Returns "" once + it is in.""" - onnx, config = piper_paths() + if voice_installed(name): + return "" - if not onnx.is_file() or not config.is_file(): - try: - onnx_url, config_url = _piper_urls() - except ValueError: - return ( - f"'{piper_voice()}' is not a Piper voice name. Use " - "one shaped like en_US-amy-medium." - ) + try: + onnx_url, config_url = _piper_urls(name) + except ValueError: + return ( + f"'{name}' is not a Piper voice name. Use one shaped like " + "en_US-amy-medium." + ) - why = _download(onnx_url, onnx, "voice", on_progress) - if why: - return why + onnx, config = piper_paths(name) + MODELS_DIR.mkdir(parents=True, exist_ok=True) + return (_download(onnx_url, onnx, "voice", on_progress) + or _download(config_url, config, "voice settings", on_progress)) - why = _download(config_url, config, "voice settings", on_progress) - if why: - return why - return "" +def remove_listening(name: str) -> None: + """Delete the Vosk model NAME, which must be one Settings offers.""" + + if name not in {c.name for c in LISTENING}: + raise ValueError(f"{name!r} is not a listening model Flash offers") + shutil.rmtree(vosk_model_dir(name), ignore_errors=True) + + +def remove_voice(name: str) -> None: + """Delete the Piper voice NAME, which must be one Settings offers.""" + + if name not in {c.name for c in SPEAKING}: + raise ValueError(f"{name!r} is not a voice Flash offers") + for path in piper_paths(name): + path.unlink(missing_ok=True) _listener = None @@ -568,6 +644,10 @@ def _load_speaker(): if _speaker is None or _speaker[0] != onnx: from piper import PiperVoice + # A sound the voice lacks is dropped either way; Piper saying so + # in the terminal, once per sound, only gets in the way. + logging.getLogger("piper").setLevel(logging.ERROR) + settings = json.loads(config.read_text(encoding="utf-8")) rate = int(settings.get("audio", {}).get("sample_rate", 22050)) _speaker = (onnx, PiperVoice.load(str(onnx)), rate) @@ -575,9 +655,49 @@ def _load_speaker(): return _speaker[1], _speaker[2] +# Symbols a voice cannot say as written, with the words it can. Longer +# ones first, so "->" is not read as "-" then ">". +_SAID_AS = ( + ("->", " to "), ("=>", " to "), ("\u2192", " to "), ("\u21d2", " to "), + (">=", " at least "), ("\u2265", " at least "), + ("<=", " at most "), ("\u2264", " at most "), + ("!=", " is not "), ("\u2260", " is not "), ("==", " equals "), + ("&&", " and "), ("||", " or "), ("&", " and "), + ("\u00b1", " plus or minus "), ("\u00d7", " times "), + ("\u00f7", " divided by "), ("\u00b0", " degrees "), + ("\u2044", " over "), + ("+", " plus "), ("=", " equals "), ("@", " at "), +) +# Marks from code and Markdown, said as a pause, not spelled out. +# Brackets above all: Piper takes [[ ... ]] as raw phonemes, and reads +# whatever is inside as sounds its voice may not have. +_UNSAID = re.compile(r"[\[\]{}<>|`*#^~_\\\u2022\u2023\u25e6]+") + + +def fit_for_voice(text: str) -> str: + """TEXT in words a voice can say. + + Code, paths, and symbols come through otherwise as sounds the voice + has none of: Piper drops them with a warning, and what is left comes + out garbled. + """ + + # Compatibility forms first: ² as 2, fi as fi, a full-width letter + # as itself. + text = unicodedata.normalize("NFKC", text or "") + for mark, words in _SAID_AS: + text = text.replace(mark, words) + text = _UNSAID.sub(" ", text) + return " ".join(text.split()) + + def _pcm_chunks(voice, text: str): """Yield raw 16-bit audio for `text`, across Piper's two APIs.""" + text = fit_for_voice(text) + if not text: + return + if hasattr(voice, "synthesize_stream_raw"): yield from voice.synthesize_stream_raw(text) return @@ -655,6 +775,105 @@ def _abort(stream) -> None: pass +# The page asks for both from threads of its own; loading a model twice +# at once would only waste the memory of one, and Piper is not promised +# to be safe to call from two threads together. +_models_lock = threading.Lock() + + +def transcribe(pcm: bytes) -> tuple[str, str]: + """What was said in PCM: 16 kHz mono 16-bit audio, as the web page + records it. Returns `(text, "")`, or `("", reason)` when the listening + model cannot be used.""" + + try: + import vosk + except Exception: # noqa: BLE001 + return "", INSTALL_HINT + + try: + with _models_lock: + model = _load_listener() + except Exception as exc: # noqa: BLE001 + return "", f"could not load the listening model: {exc}" + + recognizer = vosk.KaldiRecognizer(model, SAMPLE_RATE) + block = BLOCK_FRAMES * 2 + for start in range(0, len(pcm) - len(pcm) % 2, block): + recognizer.AcceptWaveform(pcm[start:start + block]) + + try: + result = json.loads(recognizer.FinalResult()) + except ValueError: + return "", "the listening model returned nothing usable." + + return str(result.get("text") or "").strip(), "" + + +def synthesize(text: str) -> tuple[bytes, str]: + """TEXT spoken, as a WAV file the web page can play. Returns + `(wav, "")`, or `(b"", reason)` when the voice cannot be used.""" + + text = (text or "").strip() + if not text: + return b"", "" + + try: + with _models_lock: + voice, rate = _load_speaker() + pcm = b"".join(chunk or b"" for chunk in _pcm_chunks(voice, text)) + except ImportError: + return b"", INSTALL_HINT + except Exception as exc: # noqa: BLE001 + return b"", f"could not speak the reply: {exc}" + + out = io.BytesIO() + with wave.open(out, "wb") as wav: + wav.setnchannels(1) + wav.setsampwidth(2) + wav.setframerate(rate) + wav.writeframes(pcm) + return out.getvalue(), "" + + +def speakable(line: str) -> str: + """One line of a reply, as the page reads it off the screen, made fit + to hear: an address is said as "a link", not spelled out, and emojis + are left unsaid.""" + + return " ".join(_EMOJI.sub("", _URL.sub("a link", line or "")).split()) + + +# Said to a paused session, these wake it; said as a whole turn, these +# pause it. Matched on the words alone, like the exit phrases. +RESUME_WORD = "resume" +PAUSE_PHRASES = {"pause", "pause voice", "pause voice mode", "hold on"} + + +def _words(text: str) -> list[str]: + return re.sub(r"[^a-z' ]", "", (text or "").lower()).split() + + +def voice_command(text: str, speaking: str = "") -> dict: + """What a turn of speech asks of voice mode, beyond being a message. + + `speaking` is the line Flash was reading out when it was heard: said + over a reply, "interrupt" cuts it off, but not when the word came + from the reply itself, echoing back through the microphone. + """ + + words = _words(text) + return { + "exit": is_exit_phrase(text), + "pause": " ".join(words) in PAUSE_PHRASES, + "resume": RESUME_WORD in words, + "interrupt": _heard_interruption(text) + and not _heard_interruption(speaking), + # Only the interrupt word, with nothing playing to cut off. + "alone": " ".join(words) == interrupt_word(), + } + + CODE_ONLY = "That reply is code. It is on screen." CUT_SHORT = "The rest is on screen." @@ -662,6 +881,12 @@ def _abort(stream) -> None: _IMAGE = re.compile(r"!\[[^\]]*\]\([^)]*\)") _LINK = re.compile(r"\[([^\]]+)\]\([^)]*\)") _URL = re.compile(r"?") +# Pictographs, dingbats, flags, and the joiners and selectors that knit +# them together: seen on screen, never read out. +_EMOJI = re.compile( + "[\U0001F000-\U0001FAFF\u2600-\u27BF\u2B00-\u2BFF" + "\uFE0E\uFE0F\u200D\u20E3\U000E0020-\U000E007F]" +) _TABLE = re.compile(r"^\s*\|.*$", re.MULTILINE) _RULE = re.compile(r"^\s*([-*_]\s*){3,}$", re.MULTILINE) _HEADING = re.compile(r"^\s{0,3}#{1,6}\s*", re.MULTILINE) @@ -685,7 +910,7 @@ def for_speech(text: str, limit: int = 0) -> str: # the reader can finish on screen. limit = limit or max_speech_chars() - stripped = _FENCE.sub(" ", text or "") + stripped = _EMOJI.sub("", _FENCE.sub(" ", text or "")) stripped = _IMAGE.sub(" ", stripped) stripped = _LINK.sub(r"\1", stripped) stripped = _URL.sub("a link", stripped) diff --git a/flash/web.py b/flash/web.py index fc696e2..7aba5f6 100644 --- a/flash/web.py +++ b/flash/web.py @@ -57,6 +57,7 @@ memory, skills, updater, + voice, workspace, ) from .dashes import DashGuard @@ -118,6 +119,11 @@ MAX_BODY_BYTES = 1_000_000 # An upload arrives as base64 in JSON: a third bigger than the file. MAX_UPLOAD_BODY = workspace.MAX_UPLOAD_BYTES * 4 // 3 + 64_000 +# Voice from the page arrives as base64 16 kHz 16-bit mono audio, up to +# the longest turn the terminal's voice mode would record. +MAX_VOICE_BODY = ( + int(voice.MAX_TURN_SECONDS) * voice.SAMPLE_RATE * 2 * 4 // 3 + 64_000 +) # The most attachments one message carries. MAX_ATTACHMENTS = 10 @@ -238,6 +244,9 @@ class Chat: # running turn's next step; a "queue" one becomes the next turn. pending: list[dict] = field(default_factory=list) stop: threading.Event = field(default_factory=threading.Event) + # The turn running now came from voice mode: its reply is heard, so + # the model is asked to keep it short and plain. + heard: bool = False created: float = field(default_factory=time.time) updated: float = field(default_factory=time.time) project: str = "" @@ -621,6 +630,10 @@ class Session: def __init__(self, hub: Optional[Hub] = None) -> None: self.hub = hub or Hub() self.lan = False + # The voice models are being downloaded for the page: the first + # use's pair, or one model picked in Settings, (kind, name). + self.voice_setup = False + self.voice_job: Optional[tuple[str, str]] = None # The link's token and the browsers signed in with it. A server # restarted after an update takes over the old one's (Server). self.access = Access() @@ -732,6 +745,82 @@ def state(self, lite: bool = False) -> dict: }, } + # Voice --------------------------------------------------------- + + def set_up_voice(self) -> None: + """Download the voice models, once, telling every page how far + along it is.""" + + with self._lock: + if self.voice_setup or self.voice_job: + return + self.voice_setup = True + + def run() -> None: + said: dict[str, int] = {} + + def progress(label: str, percent: int) -> None: + if said.get(label) != percent: + said[label] = percent + self.hub.publish({ + "type": "voice-setup", "label": label, + "percent": percent, + }) + + why = "" + try: + why = voice.ensure_models(progress) + finally: + with self._lock: + self.voice_setup = False + self.hub.publish( + {"type": "voice-setup", "done": True, "error": why} + ) + + threading.Thread(target=run, daemon=True).start() + + def fetch_voice_model(self, kind: str, name: str) -> None: + """Download one voice model picked in Settings and, once it is + in, make it the one in use. Pages follow along through + "voice-model" events.""" + + from . import ai # deferred: ai imports half of Flash + + with self._lock: + if self.voice_setup or self.voice_job: + raise ValueError("A voice model is already downloading.") + self.voice_job = (kind, name) + + def run() -> None: + said: list[int] = [-1] + + def progress(label: str, percent: int) -> None: + if said[0] != percent: + said[0] = percent + self.hub.publish({ + "type": "voice-model", "kind": kind, "name": name, + "label": label, "percent": percent, + }) + + why = "" + try: + fetch = (voice.download_listening if kind == "listening" + else voice.download_voice) + why = fetch(name, progress) + if not why: + ai.set_config_var(VOICE_SETTINGS[kind], name) + except Exception as exc: # noqa: BLE001 + why = str(exc) + finally: + with self._lock: + self.voice_job = None + self.hub.publish({ + "type": "voice-model", "kind": kind, "name": name, + "done": True, "error": why, + }) + + threading.Thread(target=run, daemon=True).start() + # Sub-agents ---------------------------------------------------- def adopt(self, chat: Chat, agent_ids: set) -> None: @@ -903,7 +992,7 @@ def answer(self, ask_id: str, answer: str) -> bool: def send( self, chat: Chat, text: str, wake: bool = False, mode: str = "", - files: Optional[list] = None, + files: Optional[list] = None, heard: bool = False, ) -> dict: """Start a turn on a thread of its own and return at once. @@ -934,6 +1023,7 @@ def send( else: chat.stop.clear() chat.queued = True + chat.heard = heard waiting = False if waiting: @@ -1314,7 +1404,7 @@ def run_turn( with capture_tool_output(_sink(session, chat)), \ answer_from(session.answerer(chat)): ai._fit_and_compact(ai.console, client, chat.messages) - prompt = ai._session_system_prompt() + prompt = ai._session_system_prompt(heard=chat.heard) if found is not None: prompt = f"{prompt}\n\n{project_prompt(found)}".strip() system = ai._message("system", prompt) @@ -1627,6 +1717,44 @@ def scene_data(name: str) -> Optional[dict]: } +# The two kinds of voice model Settings manages: which catalogue, how to +# tell one is in, which setting names the one in use, how to remove one. +VOICE_KINDS = { + "listening": ("LISTENING", "listening_installed", "vosk_model", + "remove_listening"), + "speaking": ("SPEAKING", "voice_installed", "piper_voice", + "remove_voice"), +} +VOICE_SETTINGS = { + "listening": "VOICE_VOSK_MODEL", "speaking": "VOICE_PIPER_VOICE", +} + + +def voice_models(session: "Session") -> dict: + """Every voice model Settings offers, and where each one stands.""" + + def listed(kind: str) -> list[dict]: + offered, installed, setting, _ = VOICE_KINDS[kind] + in_use = getattr(voice, setting)() + return [ + { + "name": c.name, "size": c.size, "label": c.label, + "installed": getattr(voice, installed)(c.name), + "current": c.name == in_use, + } + for c in getattr(voice, offered) + ] + + job = session.voice_job + return { + "listening": listed("listening"), + "speaking": listed("speaking"), + "downloading": {"kind": job[0], "name": job[1]} if job else None, + "missing": voice.web_missing(), + "install": voice.INSTALL_HINT, + } + + def _extension_info(ext: "extensions.Extension") -> dict: return { "name": ext.name, @@ -1867,6 +1995,42 @@ def command(session: Session, body: dict, browser: str = "") -> dict: session.relink() return {"signed_out": len(gone), "you": bool(browser in gone)} + if name == "voice-status": + return { + "missing": voice.web_missing(), + "ready": voice.models_present(), + "install": voice.INSTALL_HINT, + } + + if name == "voice-setup": + # The models download once, on a thread; the page follows along + # through "voice-setup" events. + if voice.models_present(): + return {"ready": True} + session.set_up_voice() + return {"ready": False} + + if name == "voice-models": + return voice_models(session) + + if name in ("voice-model", "voice-model-remove"): + kind = str(body.get("kind") or "") + if kind not in VOICE_KINDS: + raise ValueError("pick listening or speaking") + offered, installed, setting, remove = VOICE_KINDS[kind] + if arg not in {c.name for c in getattr(voice, offered)}: + raise ValueError(f"{arg!r} is not a model Flash offers") + if name == "voice-model-remove": + if arg == getattr(voice, setting)(): + raise ValueError("That one is in use. Pick another first.") + getattr(voice, remove)(arg) + return voice_models(session) + if getattr(voice, installed)(arg): + ai.set_config_var(VOICE_SETTINGS[kind], arg) + return voice_models(session) + session.fetch_voice_model(kind, arg) + return voice_models(session) + if name == "lan": if session.switch_lan is None: raise ValueError("This Flash cannot reopen its server.") @@ -2250,9 +2414,10 @@ def do_POST(self) -> None: return length = int(self.headers.get("Content-Length") or 0) - limit = ( - MAX_UPLOAD_BODY if url.path == "/api/upload" else MAX_BODY_BYTES - ) + limit = { + "/api/upload": MAX_UPLOAD_BODY, + "/api/voice/hear": MAX_VOICE_BODY, + }.get(url.path, MAX_BODY_BYTES) if length > limit: self._json( {"error": "too large"}, HTTPStatus.REQUEST_ENTITY_TOO_LARGE @@ -2281,6 +2446,7 @@ def _post(self, path: str, body: dict) -> dict: chat, str(body.get("text", "")), mode=str(body.get("mode") or ""), files=[str(f) for f in body.get("files") or []], + heard=bool(body.get("voice")), ) return {"chat": chat.id, **sent} @@ -2293,6 +2459,31 @@ def _post(self, path: str, body: dict) -> dict: raise ValueError("that upload did not arrive whole") from None return workspace.keep_upload(str(body.get("name") or ""), data) + if path == "/api/voice/hear": + try: + pcm = base64.b64decode( + str(body.get("data") or ""), validate=True + ) + except (ValueError, TypeError): + raise ValueError( + "that recording did not arrive whole" + ) from None + text, why = voice.transcribe(pcm) + if why: + raise ValueError(why) + return { + "text": text, + **voice.voice_command(text, str(body.get("speaking") or "")), + } + + if path == "/api/voice/say": + wav, why = voice.synthesize( + voice.speakable(str(body.get("text") or "")) + ) + if why: + raise ValueError(why) + return {"audio": base64.b64encode(wav).decode("ascii")} + if path == "/api/answer": return { "ok": session.answer( diff --git a/flash/web/index.html b/flash/web/index.html index 4dd5b27..a7250b5 100644 --- a/flash/web/index.html +++ b/flash/web/index.html @@ -442,6 +442,49 @@ } #jump svg { width: 16px; height: 16px; } +/* Voice mode: the box gives way to an orb that listens and speaks. */ +body.voice-on .shell, body.voice-on #suggestions, body.voice-on #pending { display: none !important; } +#voice { display: flex; align-items: center; gap: 16px; padding: 14px 14px 14px 16px; border: 1px solid var(--border-strong); border-radius: 22px; background: var(--surface); box-shadow: var(--composer-shadow); animation: pop 0.2s var(--ease); } +#voice[hidden] { display: none; } +.voice-orb { --lvl: 0; position: relative; flex: none; width: 58px; height: 58px; border-radius: 50%; cursor: pointer; + background: radial-gradient(circle at 34% 30%, #ffe1bd 0%, #f0a36f 30%, var(--accent) 62%, color-mix(in srgb, var(--accent) 55%, #000) 100%); + box-shadow: 0 0 calc(8px + var(--lvl) * 44px) color-mix(in srgb, var(--accent) 55%, transparent); + transform: scale(calc(1 + var(--lvl) * 0.22)); transition: transform 0.08s linear, box-shadow 0.08s linear, filter 0.3s; } +.voice-orb span { position: absolute; inset: -7px; border-radius: 50%; border: 2px solid color-mix(in srgb, var(--accent) 45%, transparent); opacity: 0; } +#voice[data-state="thinking"] .voice-orb, #voice[data-state="hearing"] .voice-orb, #voice[data-state="setup"] .voice-orb { animation: orb-breathe 1.5s ease-in-out infinite; } +#voice[data-state="listening"] .voice-orb span, #voice[data-state="asking"] .voice-orb span { animation: orb-ring 1.8s ease-out infinite; } +#voice[data-state="paused"] .voice-orb { filter: saturate(0.25) brightness(0.8); } +@keyframes orb-breathe { 50% { transform: scale(0.9); filter: brightness(0.85); } } +@keyframes orb-ring { from { opacity: 0.8; transform: scale(0.9); } to { opacity: 0; transform: scale(1.35); } } +.voice-say { flex: 1; min-width: 0; display: flex; flex-direction: column; gap: 2px; } +.voice-say b { font-size: 15.5px; font-weight: 600; color: var(--text); } +.voice-say span { font-size: 13px; color: var(--text-3); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.voice-end { flex: none; } +@media (prefers-reduced-motion: reduce) { .voice-orb { transform: none; } } + +/* Settings, voice models. */ +.voice-models { max-width: 620px; } +.vm-missing { font: 12.5px var(--mono); color: var(--warn); margin: 0 0 12px; } +.vm-head { display: flex; align-items: baseline; gap: 10px; margin: 16px 0 8px; } +.vm-head:first-child, .vm-missing + .vm-head { margin-top: 0; } +.vm-head b { font-size: 14px; font-weight: 600; color: var(--text); } +.vm-head span { font-size: 13px; color: var(--text-3); } +.vm-list { border: 1px solid var(--border); border-radius: 14px; background: var(--surface); overflow: hidden; } +.vm-row { display: flex; align-items: center; gap: 12px; padding: 11px 12px 11px 14px; } +.vm-row + .vm-row { border-top: 1px solid var(--border); } +.vm-mark { width: 16px; flex: none; color: var(--accent); display: grid; place-items: center; } +.vm-mark svg { width: 16px; height: 16px; } +.vm-what { flex: 1; min-width: 0; display: flex; flex-direction: column; gap: 2px; } +.vm-what b { font-size: 14px; font-weight: 500; color: var(--text); } +.vm-dot { color: var(--text-3); margin: 0 3px; font-size: 11px; vertical-align: 1px; } +.vm-what small { font: 12px var(--mono); color: var(--text-3); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.vm-act { flex: none; display: flex; align-items: center; gap: 6px; } +.vm-act .btn.small { padding: 5px 12px; font-size: 13px; } +.vm-act .btn:disabled { opacity: 0.5; cursor: default; } +.vm-tag { font-size: 12px; font-weight: 500; color: var(--accent); background: color-mix(in srgb, var(--accent) 12%, transparent); padding: 3px 9px; border-radius: 999px; } +.vm-progress { font: 12.5px var(--mono); color: var(--text-2); } +.vm-row.current .vm-what b { color: var(--text); } + /* Composer ------------------------------------------------------------- */ #dock { flex: none; padding: 0 0 max(18px, env(safe-area-inset-bottom)); position: relative; } #composer { @@ -1058,6 +1101,8 @@ /* The message a search led to glows, and its words are marked. */ ::highlight(found) { background-color: color-mix(in srgb, var(--accent) 38%, transparent); color: var(--text); } +/* What voice mode is reading aloud right now. */ +::highlight(speaking) { background-color: color-mix(in srgb, var(--accent) 26%, transparent); color: var(--text); } .block.found { animation: found 2.2s var(--ease); border-radius: 14px; } @keyframes found { 0%, 30% { box-shadow: 0 0 0 8px color-mix(in srgb, var(--accent) 16%, transparent); background: color-mix(in srgb, var(--accent) 7%, transparent); } @@ -1223,6 +1268,11 @@