From c47402201c7f04b32cc88444b727262dd0b268a9 Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 10:35:16 -0400 Subject: [PATCH 1/7] feat(speak): add --play and --list-voices to the speak command MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two new-user quality-of-life features on `dg speak`: - `--list-voices`: discover available TTS voices from the CLI (mirrors the `dg models` table), with an aura/flux badge derived from the model-name prefix, and the default voice named in the footer. Renders only in default output mode so `-o json` stays parseable; exits before any TTS generation. - `--play`: play the audio through the first available system player (ffplay, afplay, paplay, aplay) instead of hand-piping to ffplay. Fails fast with a clear message when no player is installed, and when the chosen player cannot decode the requested format (raw mulaw/alaw with no container; mp3 on paplay/aplay). Works together with -o (save AND play), and satisfies the "where does the audio go" requirement so it is allowed on a TTY. Now that Flux TTS (flux-alexis-en) is the default (#89), the default playback path is the Speak v2 linear16 stream. It is fed to the player through the existing `_streaming_wav_header` helper as frames arrive, so playback starts at first-audio latency and needs no extra player flags — the same framing `dg speak | ffplay -` already relies on. With -o the saved file still gets an exact-length header via `_pcm_to_wav`, written before the wait on the player so an interrupted playback still leaves the file behind. afplay (no stdin mode) is handed a complete temp file instead. The TTY no-output guard and its message now mention --play. Docs: README, the llms.txt/llms-full.txt command reference, and the skill_generator quickstart lead with `dg speak "Hello from Deepgram" --play` and show --list-voices; the llms-full example that still saved Flux audio as .mp3 is corrected to .wav. Tests: --list-voices (default table, json mode, empty, API error), --play across the Flux streaming, Flux+afplay, Aura, and save-and-play paths, the no-player and unplayable-format errors, player exit codes, mid-stream failure, and the player-quit broken pipe. Playback plumbing is verified end to end against a stub player binary: the test asserts the exact byte stream the player receives is a valid WAV. Real audio output still needs manual verification. Co-Authored-By: Claude Opus 5 --- README.md | 14 + .../src/deepctl_cmd_speak/command.py | 457 +++++++++- .../src/deepctl_cmd_speak/models.py | 13 + .../tests/unit/test_speak_command.py | 856 +++++++++++++++++- .../src/deepctl_core/skill_generator.py | 5 +- web/public/llms-full.txt | 10 +- web/public/llms.txt | 2 +- 7 files changed, 1330 insertions(+), 27 deletions(-) diff --git a/README.md b/README.md index 3d30eac..ebe14f9 100644 --- a/README.md +++ b/README.md @@ -184,8 +184,15 @@ container so it is directly playable. Pass an `aura-*` model to use the Speak v1 batch REST API instead, which supports containerized formats like MP3. ```bash +# Quickstart — synthesize and hear it straight away +dg speak "Hello from Deepgram" --play + +# Discover voices +dg speak --list-voices + # Flux TTS (v2, WebSocket streaming) — the default dg speak "Hello from Flux" -o hello.wav +dg speak "Save it and play it" -o hello.wav --play # Piped audio is a streaming WAV; -loglevel error hides ffmpeg's cosmetic # end-of-stream notice (the audio is complete). dg speak "Hello from Flux" | ffplay -loglevel error -nodisp -autoexit - @@ -204,6 +211,13 @@ echo "Hello" | dg speak -o greeting.mp3 -m aura-2-asteria-en dg speak "Hola, bienvenido a Deepgram" -o hola.mp3 -m aura-2-selena-es ``` +`--play` uses the first available system player (`ffplay`, `afplay`, `paplay`, +or `aplay`); install `ffmpeg` if none is present. With the Flux default the +audio is streamed into the player as it arrives, so playback starts at +first-audio latency rather than after the whole utterance. `paplay` and `aplay` +decode PCM/WAV only, so playing Aura's MP3 output needs `ffplay` (or `afplay` +on macOS). + ### Text Intelligence Analyze text for sentiment, summaries, topics, and intents. diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 705a32b..736cbf1 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -3,12 +3,16 @@ from __future__ import annotations import io +import shutil import struct +import subprocess import sys +import tempfile import time import wave +from contextlib import contextmanager, suppress from pathlib import Path -from typing import TYPE_CHECKING, Any +from typing import IO, TYPE_CHECKING, Any import click from deepctl_core import ( @@ -17,15 +21,24 @@ BaseResult, Config, DeepgramClient, + get_output_format, ) from rich.console import Console +from rich.table import Table -from .models import SpeakResult +from .models import SpeakResult, SpeakVoicesResult, VoiceInfo if TYPE_CHECKING: from collections.abc import Iterator console = Console(stderr=True) +# Tables and other stdout-bound rendering (only used when no audio goes to +# stdout, i.e. --list-voices). +stdout_console = Console() + +# Keep in sync with the --model option default below; also reported by +# --list-voices so the table says which voice you get for free. +_DEFAULT_MODEL = "flux-alexis-en" # Flux (Speak v2) streaming controls, per the /v2/speak API. `speed` is a # 0.05-increment multiplier and `expressivity` is a small integer range; both @@ -34,6 +47,168 @@ _FLUX_SPEEDS = (0.85, 0.90, 0.95, 1.00, 1.05, 1.10, 1.15) _FLUX_EXPRESSIVITY = (-2, -1, 0, 1, 2) +# Audio players probed for --play, in preference order. ffplay ships with +# ffmpeg and plays every format we can emit; the rest are OS-native fallbacks. +_AUDIO_PLAYERS = ("ffplay", "afplay", "paplay", "aplay") + +# Players that read audio from stdin, and the argv that makes them do it. +# afplay (macOS) has no stdin mode, so it is handed a temp file instead. +_STDIN_PLAYER_ARGV = { + "ffplay": ["ffplay", "-loglevel", "error", "-nodisp", "-autoexit", "-"], + "paplay": ["paplay"], + "aplay": ["aplay", "-q", "-"], +} + +# Players with no stdin mode, which are handed a temp file by path instead. +_FILE_PLAYER_ARGV = {"afplay": ["afplay"]} + +# paplay and aplay decode PCM/WAV only -- they cannot play a compressed +# container such as Aura's default mp3. +_PCM_ONLY_PLAYERS = ("paplay", "aplay") + +# Compressed, self-describing encodings: a player can sniff these from the +# byte stream. The PCM encodings cannot be sniffed — Speak v1 wraps them in a +# WAV container unless `--container none` is passed, and Flux wraps its +# linear16 stream itself. +_COMPRESSED_ENCODINGS = ("mp3", "aac", "opus", "flac") +_PCM_ENCODINGS = ("linear16", "mulaw", "alaw") + +# Encoding/container -> temp-file suffix, so the temp file handed to afplay is +# sniffed correctly by CoreAudio. +_SUFFIX_BY_ENCODING = { + "mp3": ".mp3", + "linear16": ".wav", + "flac": ".flac", + "opus": ".ogg", + "aac": ".aac", +} + + +def _find_audio_player() -> str | None: + """Return the first available audio player command, or ``None``.""" + for player in _AUDIO_PLAYERS: + if shutil.which(player): + return player + return None + + +def _voice_type_badge(model_name: str) -> str: + """Derive an aura/flux badge from a TTS model name prefix.""" + name = model_name.lower() + if name.startswith("aura"): + return "aura" + if name.startswith("flux"): + return "flux" + return "tts" + + +def _play_suffix(*, is_flux: bool, encoding: str | None, container: str | None) -> str: + """Best-effort file suffix for the audio handed to afplay's temp file.""" + if is_flux: + # Flux streams raw audio; only linear16 gets a WAV wrapper (below). + return ".wav" if (encoding or "linear16") == "linear16" else ".raw" + if container == "wav": + return ".wav" + if container == "ogg": + return ".ogg" + eff_encoding = (encoding or "mp3").lower() + if eff_encoding in _PCM_ENCODINGS: + # Speak v1 defaults these to a WAV container; `--container none` is + # the only way to get them bare. + return ".raw" if container == "none" else ".wav" + return _SUFFIX_BY_ENCODING.get(eff_encoding, ".mp3") + + +def _check_playable( + player: str, *, is_flux: bool, encoding: str | None, container: str | None +) -> str | None: + """Return why ``player`` cannot play this audio format, or ``None``. + + Checked before the API call so a format the chosen player cannot decode + fails with an explanation instead of silence or a decoder error. + """ + eff_encoding = (encoding or ("linear16" if is_flux else "mp3")).lower() + compressed = eff_encoding in _COMPRESSED_ENCODINGS + + # Raw PCM with no container is undetectable: the player has no way to know + # the sample rate, width, or encoding. Flux linear16 gets a WAV wrapper + # (so it is fine); Flux mulaw/alaw and an explicit `--container none` do + # not. + raw = (is_flux and eff_encoding != "linear16") or ( + not is_flux and not compressed and container == "none" + ) + if raw: + return ( + f"--play cannot play raw {eff_encoding} audio: it has no container " + "for the player to detect. Use the default linear16 audio, add " + "--container wav (Aura), or save it with -o." + ) + + if compressed and player in _PCM_ONLY_PLAYERS: + return ( + f"'{player}' can only play PCM/WAV audio, not {eff_encoding}. " + "Install ffmpeg (ffplay) to play it, or save it with -o." + ) + return None + + +@contextmanager +def _player_stdin(player: str) -> Iterator[IO[bytes]]: + """Spawn a stdin-reading player and yield its stdin for streaming writes. + + On a clean exit the pipe is closed (which is how the player learns the + audio ended) and the process is waited for, so the command does not return + before playback finishes. If the body raises, the player is killed rather + than left holding a half-written stream. + """ + proc = subprocess.Popen(_STDIN_PLAYER_ARGV[player], stdin=subprocess.PIPE) + stdin = proc.stdin + assert stdin is not None # stdin=PIPE above + try: + yield stdin + except BaseException: + proc.kill() + proc.wait() + raise + finally: + # A player that quit early (ffplay's "q") leaves a broken pipe; that + # is the user stopping playback, not a failure. + with suppress(BrokenPipeError, OSError): + stdin.close() + returncode = proc.wait() + if returncode: + raise click.ClickException( + f"Audio player '{player}' exited with status {returncode}." + ) + + +def _play_audio(player: str, audio: bytes, *, suffix: str) -> None: + """Play a fully-synthesized blob through the detected system player. + + ffplay/paplay/aplay read it from stdin; afplay (macOS) has no stdin mode, + so the bytes go to a temp file that is played by path. Playback failures + raise so the command exits non-zero rather than reporting a success it + did not deliver. + """ + if player in _FILE_PLAYER_ARGV: + with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp: + tmp.write(audio) + tmp_path = tmp.name + try: + returncode = subprocess.run( + [*_FILE_PLAYER_ARGV[player], tmp_path], check=False + ).returncode + finally: + Path(tmp_path).unlink(missing_ok=True) + if returncode: + raise click.ClickException( + f"Audio player '{player}' exited with status {returncode}." + ) + return + + with _player_stdin(player) as sink, suppress(BrokenPipeError): + sink.write(audio) + def _fmt_bytes(n: int) -> str: """Human-readable byte count for progress display.""" @@ -173,8 +348,11 @@ class SpeakCommand(BaseCommand): # raw audio wrapped in WAV. Piped audio is a streaming WAV (unknown # length up front), so pass `-loglevel error` to silence ffmpeg's # cosmetic end-of-stream notice. + 'dg speak "Hello from Deepgram" --play', + "dg speak --list-voices", 'dg speak "Hello world"', 'dg speak "Hello world" -o hello.wav', + 'dg speak "Hello world" -o hello.wav --play', "dg speak --file message.txt -o output.wav", 'dg speak "Hello" | ffplay -loglevel error -nodisp -autoexit -', # Flux TTS streaming controls: --speed (0.85-1.15) and beta @@ -200,7 +378,9 @@ class SpeakCommand(BaseCommand): "formats like mp3. Supports model selection and audio format options. " "Flux TTS models also accept --speed (0.85–1.15) and beta " "--expressivity (-2..2; default 0 = nominal) streaming controls; these " - "are rejected for other models." + "are rejected for other models. --play sends the audio to a local " + "system player instead of (or as well as) a file, and --list-voices " + "prints the available TTS voices without generating speech." ) def get_arguments(self) -> list[dict[str, Any]]: @@ -222,11 +402,12 @@ def get_arguments(self) -> list[dict[str, Any]]: "help": ( "TTS model. flux-* = Flux TTS / Speak v2 (WebSocket " "streaming; default flux-alexis-en); aura-* = Speak v1 " - "(REST batch, e.g. aura-2-asteria-en)." + "(REST batch, e.g. aura-2-asteria-en). " + "See voices: dg speak --list-voices" ), "type": str, "is_option": True, - "default": "flux-alexis-en", + "default": _DEFAULT_MODEL, }, { "names": ["--encoding"], @@ -278,6 +459,21 @@ def get_arguments(self) -> list[dict[str, Any]]: "type": str, "is_option": True, }, + { + "names": ["--play"], + "help": ( + "Play the audio through your system player (ffplay, afplay, " + "paplay, or aplay). Can be combined with -o to save and play." + ), + "is_flag": True, + "is_option": True, + }, + { + "names": ["--list-voices"], + "help": "List available TTS voices and exit (no speech generated)", + "is_flag": True, + "is_option": True, + }, ] def handle( @@ -289,13 +485,19 @@ def handle( ) -> BaseResult | None: text = kwargs.get("text") output_path = kwargs.get("output") - model = kwargs.get("model") or "flux-alexis-en" + model = kwargs.get("model") or _DEFAULT_MODEL encoding = kwargs.get("encoding") container = kwargs.get("container") sample_rate = kwargs.get("sample_rate") speed = kwargs.get("speed") expressivity = kwargs.get("expressivity") file_path = kwargs.get("file") + play = kwargs.get("play", False) + list_voices = kwargs.get("list_voices", False) + + # --list-voices: discover voices and exit before any TTS generation. + if list_voices: + return self._list_voices(client) # Resolve text input: arg > --file > stdin if not text and file_path: @@ -314,18 +516,46 @@ def handle( message="No text provided. Pass text as argument, use --file, or pipe via stdin.", ) - # If stdout is a TTY and no output file, require --output + # If --play was requested, resolve the player up front so we fail fast + # with a clear message before spending an API call. + player: str | None = None + if play: + player = _find_audio_player() + if player is None: + return BaseResult( + status="error", + message=( + "No audio player found — install ffmpeg, or use -o to " + "save a file." + ), + ) + + # If stdout is a TTY and there's nowhere for the audio to go, require an + # explicit destination. --play satisfies that requirement. stdout_is_tty = sys.stdout.isatty() - if not output_path and stdout_is_tty: + if not output_path and not play and stdout_is_tty: return BaseResult( status="error", - message="No output specified. Use -o/--output to save to file, or pipe stdout.", + message=( + "No output specified. Use -o/--output to save to a file, " + "--play to hear it, or pipe stdout." + ), ) # Only the documented flux-* namespace uses speak.v2. Aura and unknown # model names pass through to the REST API so the service can resolve them. is_flux = model.lower().startswith("flux-") + # A player that cannot decode what we are about to request should say + # so now, before the API call, rather than emit silence or a decoder + # error after synthesis. + if player is not None: + unplayable = _check_playable( + player, is_flux=is_flux, encoding=encoding, container=container + ) + if unplayable is not None: + return BaseResult(status="error", message=unplayable) + # speed / expressivity are Flux (Speak v2) connect controls; reject them # for other models rather than silently dropping them. Raise (not return) # so the failure exits non-zero in every output mode. @@ -376,10 +606,91 @@ def handle( ) ) - if output_path: - # A WAV file must declare its data length in the header, which - # we only know once the stream ends — so buffer, then wrap and - # write. (Streaming to disk has no user-visible benefit here.) + # Flux linear16 into a stdin-reading player: stream it. The + # player gets the streaming WAV header before the first frame and + # then every frame as Flux emits it, so sound starts at + # first-audio latency instead of after the whole utterance — the + # same framing `dg speak | ffplay -` relies on. + if ( + player is not None + and player in _STDIN_PLAYER_ARGV + and eff_encoding == "linear16" + ): + console.print(f"[blue]Playing audio ({player})...[/blue]") + pcm = bytearray() + saved_bytes = 0 + try: + with _player_stdin(player) as sink: + wrote_header = False + for chunk in stream: + if not chunk: + continue + if not wrote_header: + sink.write( + _streaming_wav_header( + sample_rate=int(eff_sample_rate) + ) + ) + wrote_header = True + sink.write(chunk) + sink.flush() + if output_path: + pcm.extend(chunk) + + if output_path and pcm: + # Save before waiting on the player (that wait + # happens on leaving this block), so stopping + # playback still leaves the file behind. The file + # gets a real, exact-length header rather than the + # streaming placeholder the player was handed. + audio_bytes = _pcm_to_wav( + bytes(pcm), sample_rate=int(eff_sample_rate) + ) + Path(output_path).write_bytes(audio_bytes) + saved_bytes = len(audio_bytes) + except click.ClickException: + raise + except BrokenPipeError: + # The player exited first (ffplay's "q", for instance). + # That is the user stopping playback, not a failure. + pass + except Exception as e: + raise click.ClickException(f"Flux streaming failed: {e}") + + if prog.total == 0: + raise click.ClickException( + "Flux (Speak v2) streaming returned no audio." + ) + + if output_path: + console.print( + f"[green]Audio saved to {output_path}[/green] " + f"({saved_bytes:,} bytes — {prog.timing()})" + ) + else: + console.print( + f"[green]✓ Played {prog.total:,} bytes[/green] " + f"({prog.timing()})" + ) + return SpeakResult( + status="success", + message=( + f"Audio saved to {output_path}" + if output_path + else f"Played {prog.total:,} bytes" + ), + output_path=output_path or "", + model=model, + bytes_written=saved_bytes or prog.total, + played=True, + ) + + if output_path or player: + # Saving, or playing through afplay (which has no stdin mode), + # needs the full utterance: a WAV file must declare its data + # length in the header (known only once the stream ends), and + # afplay is handed a complete file. So buffer, then wrap once + # and reuse for both sinks. pcm = bytearray() try: for chunk in stream: @@ -403,17 +714,31 @@ def handle( audio_bytes = bytes(pcm) total_bytes = len(audio_bytes) - Path(output_path).write_bytes(audio_bytes) - console.print( - f"[green]Audio saved to {output_path}[/green] " - f"({total_bytes:,} bytes — {prog.timing()})" - ) + if output_path: + Path(output_path).write_bytes(audio_bytes) + console.print( + f"[green]Audio saved to {output_path}[/green] " + f"({total_bytes:,} bytes — {prog.timing()})" + ) + + if player: + suffix = _play_suffix( + is_flux=True, encoding=eff_encoding, container=None + ) + console.print(f"[blue]Playing audio ({player})...[/blue]") + _play_audio(player, audio_bytes, suffix=suffix) + return SpeakResult( status="success", - message=f"Audio saved to {output_path}", - output_path=output_path, + message=( + f"Audio saved to {output_path}" + if output_path + else f"Played {total_bytes:,} bytes" + ), + output_path=output_path or "", model=model, bytes_written=total_bytes, + played=bool(player), ) # Pipe path — write each chunk to stdout as it arrives so a @@ -469,7 +794,45 @@ def handle( total_bytes = 0 - if output_path: + if player: + # Playback needs the full utterance, so buffer it. When -o is + # also given, save the same bytes to the file too. + audio = bytearray() + for chunk in audio_iter: + audio.extend(chunk) + audio_bytes = bytes(audio) + total_bytes = len(audio_bytes) + + if not audio_bytes: + return BaseResult(status="error", message="TTS returned no audio.") + + if output_path: + Path(output_path).write_bytes(audio_bytes) + console.print( + f"[green]Audio saved to {output_path}[/green] " + f"({total_bytes:,} bytes)" + ) + + suffix = _play_suffix( + is_flux=False, encoding=encoding, container=container + ) + console.print(f"[blue]Playing audio ({player})...[/blue]") + _play_audio(player, audio_bytes, suffix=suffix) + + message = ( + f"Audio saved to {output_path}" + if output_path + else f"Played {total_bytes:,} bytes" + ) + return SpeakResult( + status="success", + message=message, + output_path=output_path or "", + model=model, + bytes_written=total_bytes, + played=True, + ) + elif output_path: # Write to file out = Path(output_path) with open(out, "wb") as f: @@ -511,3 +874,55 @@ def handle( # still exits 0. Reachable now that unknown models (e.g. a bare # "flux" typo) route here instead of the raising v2 path. raise click.ClickException(f"Error generating speech: {e}") + + def _list_voices(self, client: DeepgramClient) -> BaseResult: + """List available TTS voices as a table (mirrors `dg models`).""" + try: + result = client.list_models() + except Exception as e: + console.print(f"[red]Error listing voices:[/red] {e}") + return BaseResult(status="error", message=str(e)) + + voices: list[VoiceInfo] = [] + for m in result.get("tts", []): + name = m.get("name", "") + voices.append( + VoiceInfo( + name=name, + voice_type=_voice_type_badge(name), + language=m.get("language", ""), + ) + ) + + if not voices: + console.print("[yellow]No TTS voices found[/yellow]") + return SpeakVoicesResult(status="info", message="No voices found") + + # Render the human table only in default mode: for json/yaml/csv the + # framework serializes the returned result to stdout, so a table here + # would corrupt what callers pipe into jq. (No audio goes to stdout on + # this path, so the table is safe to print there, same as `dg models`.) + if get_output_format() == "default": + table = Table( + title="Deepgram TTS Voices", + show_header=True, + header_style="bold blue", + ) + table.add_column("Voice", style="green") + table.add_column("Type", style="cyan") + table.add_column("Language") + for v in voices: + table.add_row(v.name, v.voice_type, v.language) + + stdout_console.print(table) + stdout_console.print( + f"\n[dim]{len(voices)} voice(s) — generate with " + f'dg speak "..." -m ; ' + f"default: {_DEFAULT_MODEL}[/dim]" + ) + + return SpeakVoicesResult( + status="success", + voices=voices, + count=len(voices), + ) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py index 8e483db..177be57 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py @@ -3,9 +3,22 @@ from __future__ import annotations from deepctl_core import BaseResult +from pydantic import BaseModel, Field class SpeakResult(BaseResult): output_path: str = "" model: str = "" bytes_written: int = 0 + played: bool = False + + +class VoiceInfo(BaseModel): + name: str = "" + voice_type: str = "" # "aura" or "flux" + language: str = "" + + +class SpeakVoicesResult(BaseResult): + voices: list[VoiceInfo] = Field(default_factory=list) + count: int = 0 diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index f37952b..595acd2 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -1,18 +1,125 @@ """Tests for speak command.""" +import struct +import sys +import wave +from contextlib import contextmanager +from pathlib import Path from unittest.mock import Mock, patch import click import pytest from deepctl_cmd_speak.command import ( + _FILE_PLAYER_ARGV, + _STDIN_PLAYER_ARGV, SpeakCommand, + _check_playable, + _find_audio_player, _fmt_bytes, + _play_audio, + _play_suffix, + _player_stdin, + _streaming_wav_header, _StreamProgress, + _voice_type_badge, ) -from deepctl_cmd_speak.models import SpeakResult +from deepctl_cmd_speak.models import SpeakResult, SpeakVoicesResult from deepctl_core import AuthManager, BaseResult, Config, DeepgramClient +def _only(*players): + """A shutil.which side effect where only these players are installed.""" + return lambda p: f"/usr/bin/{p}" if p in players else None + + +def _wav_header_fields(blob: bytes) -> tuple[int, int, int]: + """(sample_rate, channels, bits) parsed from a 44-byte PCM WAV header.""" + assert blob[:4] == b"RIFF" + assert blob[8:12] == b"WAVE" + assert blob[12:16] == b"fmt " + channels, sample_rate = struct.unpack(" None: + self.copy_path = copy_path + self._record_path = record_path + + @property + def arg_path(self) -> str: + return self._record_path.read_text().strip() + + +@pytest.fixture +def stub_file_player(tmp_path, monkeypatch): + """Stand a real stub process in for afplay, which plays a file by path.""" + script = tmp_path / "file_player.py" + script.write_text( + "import sys\n" + "src = sys.argv[3]\n" + "with open(src, 'rb') as fh:\n" + " data = fh.read()\n" + "with open(sys.argv[1], 'wb') as fh:\n" + " fh.write(data)\n" + "with open(sys.argv[2], 'w') as fh:\n" + " fh.write(src)\n" + ) + copy_path = tmp_path / "played.bin" + record_path = tmp_path / "played_path.txt" + monkeypatch.setitem( + _FILE_PLAYER_ARGV, + "afplay", + [sys.executable, str(script), str(copy_path), str(record_path)], + ) + return _FilePlayerStub(copy_path, record_path) + + +@pytest.fixture +def failing_file_player(tmp_path, monkeypatch): + """A file-based player that exits non-zero.""" + script = tmp_path / "failing_file_player.py" + script.write_text("import sys\nsys.exit(4)\n") + monkeypatch.setitem(_FILE_PLAYER_ARGV, "afplay", [sys.executable, str(script)]) + return script + + class TestStreamProgress: """The live-progress helper used by the Flux streaming path.""" @@ -104,6 +211,8 @@ def test_get_arguments(self, command): assert "--expressivity" in option_names assert "--file" in option_names assert "-f" in option_names + assert "--play" in option_names + assert "--list-voices" in option_names @patch("deepctl_cmd_speak.command.sys") def test_handle_no_text_error( @@ -238,6 +347,8 @@ def test_handle_no_output_tty_error( assert isinstance(result, BaseResult) assert result.status == "error" assert "No output specified" in result.message + # The message names every way out, including --play. + assert "--play" in result.message @patch("deepctl_cmd_speak.command.sys") def test_handle_write_to_file( @@ -797,6 +908,749 @@ def test_handle_write_to_stdout( mock_stdout_buffer.write.assert_any_call(b"chunk2") mock_stdout_buffer.flush.assert_called_once() + # ── --list-voices ──────────────────────────────────────────────── + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="default") + def test_handle_list_voices( + self, _fmt, command, mock_config, mock_auth_manager, mock_client, capsys + ): + """--list-voices lists TTS voices with aura/flux badges, no generation.""" + mock_client.list_models.return_value = { + "stt": [{"name": "nova-3", "language": "en"}], + "tts": [ + {"name": "aura-2-asteria-en", "language": "en"}, + {"name": "flux-alexis-en", "language": "en"}, + ], + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + output=None, + model=None, + encoding=None, + container=None, + sample_rate=None, + file=None, + play=False, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.status == "success" + assert result.count == 2 + badges = {v.name: v.voice_type for v in result.voices} + assert badges["aura-2-asteria-en"] == "aura" + assert badges["flux-alexis-en"] == "flux" + # STT models are excluded; no speech is generated. + assert all("nova" not in v.name for v in result.voices) + mock_client.speak_text.assert_not_called() + mock_client.speak_text_stream.assert_not_called() + + # The table renders like `dg models`: on stdout, which carries no + # audio on this path, naming both voices and the default model. + captured = capsys.readouterr() + assert "Deepgram TTS Voices" in captured.out + assert "flux-alexis-en" in captured.out + assert "default: flux-alexis-en" in captured.out.replace("\n", "") + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") + def test_handle_list_voices_json_prints_no_table( + self, _fmt, command, mock_config, mock_auth_manager, mock_client, capsys + ): + """In json mode the framework serializes the result; no table on stdout.""" + mock_client.list_models.return_value = { + "tts": [{"name": "flux-alexis-en", "language": "en"}] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.count == 1 + captured = capsys.readouterr() + assert captured.out == "" + + def test_handle_list_voices_empty( + self, command, mock_config, mock_auth_manager, mock_client + ): + """--list-voices with no TTS voices returns an info result.""" + mock_client.list_models.return_value = {"tts": []} + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + output=None, + model=None, + encoding=None, + container=None, + sample_rate=None, + file=None, + play=False, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.status == "info" + assert result.count == 0 + + def test_handle_list_voices_api_error( + self, command, mock_config, mock_auth_manager, mock_client + ): + """A failing list_models surfaces as an error result, not a traceback.""" + mock_client.list_models.side_effect = Exception("boom") + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert result is not None + assert result.status == "error" + assert "boom" in result.message + + # ── --play ─────────────────────────────────────────────────────── + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_streams_wav_to_player( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """flux --play streams a playable WAV into the player as audio arrives.""" + mock_sys.stdin.isatty.return_value = True + # stdout is a TTY and there is no -o: --play is the destination, so the + # "no output specified" guard must not fire. + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + # The empty keep-alive chunk must not reach the player. + mock_client.speak_text_stream.return_value = iter([pcm[0], b"", pcm[1]]) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello from Flux", + output=None, + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + assert result.played is True + assert result.output_path == "" + assert result.bytes_written == 8 + + # What the player actually received: the streaming WAV header this + # command already uses for `| ffplay -`, then the PCM frames. + piped = stub_stdin_player.read_bytes() + assert piped == _streaming_wav_header(sample_rate=24000) + b"".join(pcm) + assert _wav_header_fields(piped) == (24000, 1, 16) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_and_save_together( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + tmp_path, + ): + """--play with -o plays the stream and saves an exact-length WAV.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + mock_client.speak_text_stream.return_value = iter(pcm) + output_file = tmp_path / "out.wav" + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello from Flux", + output=str(output_file), + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.played is True + assert result.output_path == str(output_file) + + # The player got the streaming framing... + assert stub_stdin_player.read_bytes() == ( + _streaming_wav_header(sample_rate=24000) + b"".join(pcm) + ) + # ...and the saved file got a real header a WAV reader can trust. + with wave.open(str(output_file), "rb") as wav: + assert wav.getframerate() == 24000 + assert wav.getnchannels() == 1 + assert wav.getsampwidth() == 2 + assert wav.readframes(wav.getnframes()) == b"".join(pcm) + + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_mulaw_to_file_is_not_wrapped( + self, + mock_sys, + command, + mock_config, + mock_auth_manager, + mock_client, + tmp_path, + ): + """Only linear16 gets a WAV wrapper; mulaw is saved as raw audio.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_client.speak_text_stream.return_value = iter([b"\xff\xfe", b"\xfd"]) + output_file = tmp_path / "out.raw" + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=str(output_file), + model="flux-alexis-en", + encoding="mulaw", + play=False, + ) + + assert isinstance(result, SpeakResult) + assert output_file.read_bytes() == b"\xff\xfe\xfd" + assert result.bytes_written == 3 + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_no_audio_fails( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """An empty Flux stream fails loudly instead of reporting playback.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([]) + + with pytest.raises(click.ClickException, match="returned no audio"): + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_stream_error_is_reported( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """A mid-stream Flux failure exits non-zero rather than half-playing.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + def exploding(): + yield b"\x01\x00" + raise RuntimeError("socket died") + + mock_client.speak_text_stream.return_value = exploding() + + with pytest.raises(click.ClickException, match="Flux streaming failed"): + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_player_exit_code_fails( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + failing_stdin_player, + ): + """A player that exits non-zero fails the command.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([b"\x01\x00\x02\x00"]) + + with pytest.raises(click.ClickException, match="exited with status 3"): + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_player_quit_is_not_an_error( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """Quitting the player mid-playback (broken pipe) is not a failure.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([b"\x01\x00", b"\x02\x00"]) + + @contextmanager + def quit_immediately(player): + sink = Mock() + sink.write.side_effect = BrokenPipeError + yield sink + + with patch("deepctl_cmd_speak.command._player_stdin", quit_immediately): + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + assert result.played is True + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_afplay_gets_a_complete_wav_file( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_file_player, + ): + """afplay has no stdin mode, so it is handed a finished WAV file.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("afplay") + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + mock_client.speak_text_stream.return_value = iter(pcm) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello from Flux", + output=None, + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.played is True + # The file afplay was pointed at is a complete, exact-length WAV. + handed_to_player = stub_file_player.copy_path + with wave.open(str(handed_to_player), "rb") as wav: + assert wav.getframerate() == 24000 + assert wav.readframes(wav.getnframes()) == b"".join(pcm) + assert str(stub_file_player.arg_path).endswith(".wav") + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_aura_play_pipes_container_bytes( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """Aura --play pipes the container bytes (mp3) straight to the player.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text.return_value = iter([b"ID3", b"mp3-data"]) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + assert result.played is True + assert result.bytes_written == 11 + assert stub_stdin_player.read_bytes() == b"ID3mp3-data" + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_aura_play_and_save_together( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + tmp_path, + ): + """--play together with -o both saves the file and plays the audio.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text.return_value = iter([b"chunk1", b"chunk2"]) + output_file = tmp_path / "out.mp3" + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=str(output_file), + model="aura-2-asteria-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.played is True + assert result.output_path == str(output_file) + assert output_file.read_bytes() == b"chunk1chunk2" + assert stub_stdin_player.read_bytes() == b"chunk1chunk2" + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_aura_play_no_audio_errors( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """An empty Aura response is an error, not a silent success.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text.return_value = iter([]) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert result is not None + assert result.status == "error" + assert "no audio" in result.message + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_no_player_found( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """--play with no available player fails clearly before any API call.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.return_value = None + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert isinstance(result, BaseResult) + assert result.status == "error" + assert "No audio player found" in result.message + mock_client.speak_text.assert_not_called() + mock_client.speak_text_stream.assert_not_called() + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_rejects_raw_flux_encoding( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """Raw mulaw has no container to detect, so --play refuses it up front.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + encoding="mulaw", + play=True, + ) + + assert result is not None + assert result.status == "error" + assert "raw mulaw" in result.message + mock_client.speak_text_stream.assert_not_called() + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_rejects_mp3_on_pcm_only_player( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """aplay cannot decode Aura's mp3, so say so instead of playing noise.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("aplay") + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert result is not None + assert result.status == "error" + assert "can only play PCM/WAV" in result.message + mock_client.speak_text.assert_not_called() + + +class TestAudioPlayerHelpers: + """The player-detection, format, and playback helpers behind --play.""" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_prefers_ffplay(self, mock_shutil): + mock_shutil.which.side_effect = lambda p: f"/usr/bin/{p}" + assert _find_audio_player() == "ffplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_falls_back_in_order(self, mock_shutil): + # ffplay missing, afplay present → afplay wins over paplay/aplay. + mock_shutil.which.side_effect = lambda p: ( + f"/usr/bin/{p}" if p in ("afplay", "paplay", "aplay") else None + ) + assert _find_audio_player() == "afplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_prefers_paplay_over_aplay(self, mock_shutil): + mock_shutil.which.side_effect = lambda p: ( + f"/usr/bin/{p}" if p in ("paplay", "aplay") else None + ) + assert _find_audio_player() == "paplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_last_resort_aplay(self, mock_shutil): + mock_shutil.which.side_effect = _only("aplay") + assert _find_audio_player() == "aplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_none(self, mock_shutil): + mock_shutil.which.return_value = None + assert _find_audio_player() is None + + def test_voice_type_badge(self): + assert _voice_type_badge("aura-2-asteria-en") == "aura" + assert _voice_type_badge("Aura-Asteria-EN") == "aura" + assert _voice_type_badge("flux-alexis-en") == "flux" + assert _voice_type_badge("something-else") == "tts" + + def test_play_suffix(self): + assert _play_suffix(is_flux=True, encoding="linear16", container=None) == ".wav" + assert _play_suffix(is_flux=True, encoding=None, container=None) == ".wav" + assert _play_suffix(is_flux=True, encoding="mulaw", container=None) == ".raw" + assert _play_suffix(is_flux=False, encoding="mp3", container=None) == ".mp3" + assert _play_suffix(is_flux=False, encoding=None, container=None) == ".mp3" + assert _play_suffix(is_flux=False, encoding=None, container="wav") == ".wav" + assert _play_suffix(is_flux=False, encoding="opus", container="ogg") == ".ogg" + assert _play_suffix(is_flux=False, encoding="flac", container=None) == ".flac" + # Speak v1 wraps PCM encodings in WAV unless --container none. + assert _play_suffix(is_flux=False, encoding="mulaw", container=None) == ".wav" + assert _play_suffix(is_flux=False, encoding="mulaw", container="none") == ".raw" + + def test_check_playable_accepts_what_players_can_decode(self): + assert ( + _check_playable("ffplay", is_flux=True, encoding=None, container=None) + is None + ) + assert ( + _check_playable("aplay", is_flux=True, encoding="linear16", container=None) + is None + ) + assert ( + _check_playable("ffplay", is_flux=False, encoding=None, container=None) + is None + ) + assert ( + _check_playable("afplay", is_flux=False, encoding="mp3", container=None) + is None + ) + + def test_check_playable_rejects_raw_audio(self): + flux_raw = _check_playable( + "ffplay", is_flux=True, encoding="alaw", container=None + ) + assert flux_raw is not None + assert "raw alaw" in flux_raw + + aura_raw = _check_playable( + "ffplay", is_flux=False, encoding="linear16", container="none" + ) + assert aura_raw is not None + assert "no container" in aura_raw + + def test_check_playable_rejects_compressed_audio_on_pcm_only_players(self): + for player in ("paplay", "aplay"): + problem = _check_playable( + player, is_flux=False, encoding="mp3", container=None + ) + assert problem is not None + assert "can only play PCM/WAV" in problem + + def test_play_audio_pipes_to_stdin_player(self, stub_stdin_player): + """The bytes handed to a stdin player are exactly the audio.""" + _play_audio("ffplay", b"audio-bytes", suffix=".wav") + + assert stub_stdin_player.read_bytes() == b"audio-bytes" + + def test_play_audio_default_ffplay_argv_reads_stdin(self): + """The real ffplay invocation is the documented no-extra-flags one.""" + assert _STDIN_PLAYER_ARGV["ffplay"] == [ + "ffplay", + "-loglevel", + "error", + "-nodisp", + "-autoexit", + "-", + ] + assert _STDIN_PLAYER_ARGV["aplay"][-1] == "-" + + def test_play_audio_raises_on_player_failure(self, failing_stdin_player): + with pytest.raises(click.ClickException, match="exited with status 3"): + _play_audio("ffplay", b"audio", suffix=".wav") + + def test_play_audio_file_player_gets_the_bytes_and_cleans_up( + self, stub_file_player + ): + """afplay is handed a temp file with the audio, removed afterwards.""" + _play_audio("afplay", b"wav-bytes", suffix=".wav") + + assert stub_file_player.copy_path.read_bytes() == b"wav-bytes" + assert str(stub_file_player.arg_path).endswith(".wav") + # The temp file itself is gone once playback finishes. + assert not Path(stub_file_player.arg_path).exists() + + def test_play_audio_file_player_raises_on_failure(self, failing_file_player): + with pytest.raises(click.ClickException, match="exited with status 4"): + _play_audio("afplay", b"wav-bytes", suffix=".wav") + + def test_player_stdin_kills_the_player_when_the_body_raises(self, tmp_path): + """A failure mid-stream must not leave the player running.""" + script = tmp_path / "sleeper.py" + script.write_text("import sys\nsys.stdin.buffer.read()\n") + with ( + patch.dict(_STDIN_PLAYER_ARGV, {"ffplay": [sys.executable, str(script)]}), + pytest.raises(RuntimeError), + _player_stdin("ffplay"), + ): + raise RuntimeError("stream died") + class TestSpeakResult: """Test cases for SpeakResult model.""" diff --git a/packages/deepctl-core/src/deepctl_core/skill_generator.py b/packages/deepctl-core/src/deepctl_core/skill_generator.py index f6a9e10..d6ade6e 100644 --- a/packages/deepctl-core/src/deepctl_core/skill_generator.py +++ b/packages/deepctl-core/src/deepctl_core/skill_generator.py @@ -591,7 +591,10 @@ def render_developer_guide( lines.append("dg login # Authenticate") lines.append("dg listen audio.wav # Transcribe a file") lines.append("dg listen --mic # Live transcription from mic") - lines.append('dg speak "Hello world" # Text-to-speech') + lines.append( + 'dg speak "Hello from Deepgram" --play # Text-to-speech, played aloud' + ) + lines.append("dg speak --list-voices # List available TTS voices") lines.append("dg projects list # List projects") lines.append("dg usage # View API usage") lines.append("dg mcp # Start MCP server") diff --git a/web/public/llms-full.txt b/web/public/llms-full.txt index 998a441..5b76e4d 100644 --- a/web/public/llms-full.txt +++ b/web/public/llms-full.txt @@ -144,10 +144,12 @@ ffmpeg -i audio.mp3 -f s16le -ar 16000 -ac 1 - | dg listen --encoding linear16 - ## Text-to-Speech ```bash +dg speak "Hello from Deepgram" --play # Synthesize and play it +dg speak --list-voices # Discover voices dg speak "Hello from Deepgram" -dg speak "Hello" --output output.mp3 -dg speak "Hello" | ffplay -nodisp -autoexit - # Pipe to player -echo "Hello world" | dg speak # Stdin input +dg speak "Hello" --output output.wav # Flux emits WAV; use -m aura-* for mp3 +dg speak "Hello" | ffplay -nodisp -autoexit - # Pipe to player +echo "Hello world" | dg speak # Stdin input dg speak --file script.txt # From file ``` @@ -161,6 +163,8 @@ dg speak --file script.txt # From file | `--encoding` | Audio encoding — Aura: mp3, linear16, flac, mulaw, alaw, opus, aac; Flux: linear16 (default), mulaw, alaw | | `--container` | Audio container — none, wav, ogg (Aura only; Flux auto-wraps linear16 in WAV) | | `--file` | Read text from file | +| `--play` | Play the audio through a local player (ffplay, afplay, paplay, aplay); combine with `--output` to save and play | +| `--list-voices` | List available TTS voices and exit, without generating speech | --- diff --git a/web/public/llms.txt b/web/public/llms.txt index c6a0a80..e144040 100644 --- a/web/public/llms.txt +++ b/web/public/llms.txt @@ -17,7 +17,7 @@ - `dg login` — Authenticate via browser (OIDC device flow) or `--api-key` flag - `dg listen ` — Transcribe files, URLs, microphone (`--mic`), or stdin. Supports `--diarize`, `--webvtt`, `--srt`, `--summarize`. (`dg transcribe` is a hidden alias) -- `dg speak ""` — Text-to-speech synthesis; streams Flux TTS by default (flux-alexis-en), or pick an Aura voice with `-m aura-2-*` +- `dg speak ""` — Text-to-speech synthesis; streams Flux TTS by default (flux-alexis-en), or pick an Aura voice with `-m aura-2-*`. `--play` hears it straight away; `--list-voices` lists the voices - `dg read ` — Text intelligence: sentiment, summaries, topics, intents - `dg models` — List available Deepgram models - `dg projects` — Manage Deepgram projects From c6da122c599b91359addf53a4f0e8115db45ad24 Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 10:40:48 -0400 Subject: [PATCH 2/7] fix(speak): keep --play honest about the file it saved and the empty pipe Two findings from review of #110: - `--play` with `-o`: the save happened inside the player-streaming block, so a player that quit first (broken pipe) skipped the write while the command still printed "Audio saved to " and returned success with that path. The broken pipe is now handled per chunk: playback stops, the stream keeps draining, and the -o file still gets the whole utterance. - `--play` with stdout redirected or piped and no `-o` collected nothing and gave no hint why. It now prints a note to stderr saying the audio went to the player and that -o saves a file. Tests: playback quit mid-stream still writes the complete WAV; the redirected stdout note; and the skill_generator quickstart lines are asserted in test_skill_generator, matching the precedent from #92. Co-Authored-By: Claude Opus 5 --- .../src/deepctl_cmd_speak/command.py | 41 +++++++--- .../tests/unit/test_speak_command.py | 80 ++++++++++++++++++- .../tests/unit/test_skill_generator.py | 3 + 3 files changed, 110 insertions(+), 14 deletions(-) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 736cbf1..5767590 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -542,6 +542,15 @@ def handle( ), ) + if play and not output_path and not stdout_is_tty: + # Playing sends the audio to the player, not to stdout, so a + # redirect or pipe would otherwise collect nothing and give no + # hint why. + console.print( + "[yellow]Note:[/yellow] --play sends the audio to your player, " + "so stdout stays empty. Add -o to save a file too." + ) + # Only the documented flux-* namespace uses speak.v2. Aura and unknown # model names pass through to the REST API so the service can resolve them. is_flux = model.lower().startswith("flux-") @@ -619,23 +628,33 @@ def handle( console.print(f"[blue]Playing audio ({player})...[/blue]") pcm = bytearray() saved_bytes = 0 + player_gone = False try: with _player_stdin(player) as sink: wrote_header = False for chunk in stream: if not chunk: continue - if not wrote_header: - sink.write( - _streaming_wav_header( - sample_rate=int(eff_sample_rate) - ) - ) - wrote_header = True - sink.write(chunk) - sink.flush() if output_path: pcm.extend(chunk) + if player_gone: + continue + try: + if not wrote_header: + sink.write( + _streaming_wav_header( + sample_rate=int(eff_sample_rate) + ) + ) + wrote_header = True + sink.write(chunk) + sink.flush() + except BrokenPipeError: + # The player exited first (ffplay's "q", for + # instance). That is the user stopping + # playback, not a failure — but keep draining + # the stream so a -o file is still complete. + player_gone = True if output_path and pcm: # Save before waiting on the player (that wait @@ -650,10 +669,6 @@ def handle( saved_bytes = len(audio_bytes) except click.ClickException: raise - except BrokenPipeError: - # The player exited first (ffplay's "q", for instance). - # That is the user stopping playback, not a failure. - pass except Exception as e: raise click.ClickException(f"Flux streaming failed: {e}") diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index 595acd2..e75d3af 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -954,7 +954,7 @@ def test_handle_list_voices( captured = capsys.readouterr() assert "Deepgram TTS Voices" in captured.out assert "flux-alexis-en" in captured.out - assert "default: flux-alexis-en" in captured.out.replace("\n", "") + assert "default: flux-alexis-en" in " ".join(captured.out.split()) @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") def test_handle_list_voices_json_prints_no_table( @@ -1276,6 +1276,84 @@ def quit_immediately(player): assert result.status == "success" assert result.played is True + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_quit_still_saves_the_whole_file( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + tmp_path, + ): + """Stopping playback must not truncate or skip the -o file.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + mock_client.speak_text_stream.return_value = iter(pcm) + output_file = tmp_path / "out.wav" + + @contextmanager + def quit_immediately(player): + sink = Mock() + sink.write.side_effect = BrokenPipeError + yield sink + + with patch("deepctl_cmd_speak.command._player_stdin", quit_immediately): + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=str(output_file), + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + # The whole utterance still reached the file the user asked for. + with wave.open(str(output_file), "rb") as wav: + assert wav.readframes(wav.getnframes()) == b"".join(pcm) + assert result.bytes_written == output_file.stat().st_size + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_with_redirected_stdout_warns( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + capsys, + ): + """Playing with stdout redirected says why the redirect gets nothing.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = False + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([b"\x01\x00"]) + + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + # Normalize rich's wrapping so the assertion is width-independent. + note = " ".join(capsys.readouterr().err.split()) + assert "--play sends the audio to your player" in note + assert "stdout stays empty" in note + @patch("deepctl_cmd_speak.command.shutil") @patch("deepctl_cmd_speak.command.sys") def test_handle_flux_play_afplay_gets_a_complete_wav_file( diff --git a/packages/deepctl-core/tests/unit/test_skill_generator.py b/packages/deepctl-core/tests/unit/test_skill_generator.py index 9074564..8b1e574 100644 --- a/packages/deepctl-core/tests/unit/test_skill_generator.py +++ b/packages/deepctl-core/tests/unit/test_skill_generator.py @@ -172,6 +172,9 @@ def test_contains_cli_section(self): assert "deepctl CLI" in content assert "dg listen" in content assert "dg login" in content + # The speak quickstart leads with playback and voice discovery. + assert 'dg speak "Hello from Deepgram" --play' in content + assert "dg speak --list-voices" in content def test_frontmatter(self): content = render_developer_guide("1.0.0", include_frontmatter=True) From 6b9dc5d6a1ac41c0e1def0f3c11ac4b27da0be1d Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 10:46:32 -0400 Subject: [PATCH 3/7] fix(speak): print the errors the framework never shows, and re-order the pipe note MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round-2 review findings on #110: - In default output mode the framework prints nothing for a returned result; it only maps the status to an exit code. Every guard in `handle()` that returns an error result was therefore exiting 1 with an empty terminal — including the two new --play ones ("No audio player found", the unplayable format message). They now go through a `_fail()` helper that puts the reason on the stderr console the way `dg models` does, and the pre-existing guards in the same function (no text, file not found, no output specified, empty Aura response) go through it too rather than staying half-silent. - The "--play leaves stdout empty" note printed before the format and speed/expressivity validation, so a rejected run led with an irrelevant note. It now prints once the request is known to be valid. Co-Authored-By: Claude Opus 5 --- .../src/deepctl_cmd_speak/command.py | 61 ++++++++++--------- .../tests/unit/test_speak_command.py | 5 ++ 2 files changed, 37 insertions(+), 29 deletions(-) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 5767590..6b1a877 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -210,6 +210,18 @@ def _play_audio(player: str, audio: bytes, *, suffix: str) -> None: sink.write(audio) +def _fail(message: str) -> BaseResult: + """An error result whose message the user actually sees. + + In default output mode the framework prints nothing for a returned + result — it only maps the status to a non-zero exit code — so the human + message has to go to the stderr console here, the way every other status + line in this command does. + """ + console.print(f"[red]Error:[/red] {message}") + return BaseResult(status="error", message=message) + + def _fmt_bytes(n: int) -> str: """Human-readable byte count for progress display.""" if n < 1024: @@ -503,17 +515,15 @@ def handle( if not text and file_path: path = Path(file_path) if not path.exists(): - return BaseResult( - status="error", message=f"File not found: {file_path}" - ) + return _fail(f"File not found: {file_path}") text = path.read_text().strip() elif not text and not sys.stdin.isatty(): text = sys.stdin.read().strip() if not text: - return BaseResult( - status="error", - message="No text provided. Pass text as argument, use --file, or pipe via stdin.", + return _fail( + "No text provided. Pass text as argument, use --file, " + "or pipe via stdin." ) # If --play was requested, resolve the player up front so we fail fast @@ -522,33 +532,17 @@ def handle( if play: player = _find_audio_player() if player is None: - return BaseResult( - status="error", - message=( - "No audio player found — install ffmpeg, or use -o to " - "save a file." - ), + return _fail( + "No audio player found — install ffmpeg, or use -o to save a file." ) # If stdout is a TTY and there's nowhere for the audio to go, require an # explicit destination. --play satisfies that requirement. stdout_is_tty = sys.stdout.isatty() if not output_path and not play and stdout_is_tty: - return BaseResult( - status="error", - message=( - "No output specified. Use -o/--output to save to a file, " - "--play to hear it, or pipe stdout." - ), - ) - - if play and not output_path and not stdout_is_tty: - # Playing sends the audio to the player, not to stdout, so a - # redirect or pipe would otherwise collect nothing and give no - # hint why. - console.print( - "[yellow]Note:[/yellow] --play sends the audio to your player, " - "so stdout stays empty. Add -o to save a file too." + return _fail( + "No output specified. Use -o/--output to save to a file, " + "--play to hear it, or pipe stdout." ) # Only the documented flux-* namespace uses speak.v2. Aura and unknown @@ -563,7 +557,7 @@ def handle( player, is_flux=is_flux, encoding=encoding, container=container ) if unplayable is not None: - return BaseResult(status="error", message=unplayable) + return _fail(unplayable) # speed / expressivity are Flux (Speak v2) connect controls; reject them # for other models rather than silently dropping them. Raise (not return) @@ -585,6 +579,15 @@ def handle( f"--expressivity must be one of: {allowed} (got {expressivity})." ) + if play and not output_path and not stdout_is_tty: + # Playing sends the audio to the player, not to stdout, so a + # redirect or pipe would otherwise collect nothing and give no + # hint why. + console.print( + "[yellow]Note:[/yellow] --play sends the audio to your player, " + "so stdout stays empty. Add -o to save a file too." + ) + if is_flux: # WebSocket streaming path. Streaming output is raw audio, so # default to linear16 @ 24kHz and wrap it in WAV for playback. @@ -819,7 +822,7 @@ def handle( total_bytes = len(audio_bytes) if not audio_bytes: - return BaseResult(status="error", message="TTS returned no audio.") + return _fail("TTS returned no audio.") if output_path: Path(output_path).write_bytes(audio_bytes) diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index e75d3af..14c8fc0 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -1504,6 +1504,7 @@ def test_handle_play_no_player_found( mock_config, mock_auth_manager, mock_client, + capsys, ): """--play with no available player fails clearly before any API call.""" mock_sys.stdin.isatty.return_value = True @@ -1525,6 +1526,10 @@ def test_handle_play_no_player_found( assert "No audio player found" in result.message mock_client.speak_text.assert_not_called() mock_client.speak_text_stream.assert_not_called() + # Default output mode prints nothing for a returned result, so the + # command has to put the reason on stderr itself. + printed = " ".join(capsys.readouterr().err.split()) + assert "No audio player found" in printed @patch("deepctl_cmd_speak.command.shutil") @patch("deepctl_cmd_speak.command.sys") From 369a6c7691f3055b2238ba0d79d70f8e5df84cdf Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 10:47:37 -0400 Subject: [PATCH 4/7] fix(speak): stop rich from eating bracketed paths out of error messages MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round-3 review finding on #110: `_fail()` interpolated the message into a rich markup string, so an error naming a path with brackets printed the path with the bracketed segment removed — `/tmp/[draft]/x.txt` came out as `/tmp//x.txt`, pointing the user at a path they never typed. The message is now escaped before printing. Co-Authored-By: Claude Opus 5 --- .../src/deepctl_cmd_speak/command.py | 7 +++-- .../tests/unit/test_speak_command.py | 27 +++++++++++++++++++ 2 files changed, 32 insertions(+), 2 deletions(-) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 6b1a877..79a81a5 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -24,6 +24,7 @@ get_output_format, ) from rich.console import Console +from rich.markup import escape from rich.table import Table from .models import SpeakResult, SpeakVoicesResult, VoiceInfo @@ -216,9 +217,11 @@ def _fail(message: str) -> BaseResult: In default output mode the framework prints nothing for a returned result — it only maps the status to a non-zero exit code — so the human message has to go to the stderr console here, the way every other status - line in this command does. + line in this command does. The message is escaped because it can carry a + user-supplied path, and rich would otherwise eat any [bracketed] segment + of it as markup. """ - console.print(f"[red]Error:[/red] {message}") + console.print(f"[red]Error:[/red] {escape(message)}") return BaseResult(status="error", message=message) diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index 14c8fc0..018f6f1 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -1531,6 +1531,33 @@ def test_handle_play_no_player_found( printed = " ".join(capsys.readouterr().err.split()) assert "No audio player found" in printed + @patch("deepctl_cmd_speak.command.sys") + def test_handle_error_message_keeps_bracketed_paths( + self, + mock_sys, + command, + mock_config, + mock_auth_manager, + mock_client, + capsys, + ): + """A path with brackets prints as typed, not half-eaten as markup.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + file="/tmp/[draft]/script.txt", + ) + + assert result is not None + assert result.status == "error" + printed = " ".join(capsys.readouterr().err.split()) + assert "/tmp/[draft]/script.txt" in printed + @patch("deepctl_cmd_speak.command.shutil") @patch("deepctl_cmd_speak.command.sys") def test_handle_play_rejects_raw_flux_encoding( From ae4e2fab3cfc20737b17aff27a6a205308377cc5 Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Wed, 23 Sep 2026 08:58:02 -0400 Subject: [PATCH 5/7] fix(speak): list voices by their canonical -m value and languages The TTS catalog carries the usable model id as `canonical_name` ("aura-2-agathe-fr") and the languages as a list. Reading only `name` and the legacy singular `language` rendered the voice as "agathe" with a blank language, so the footer's instruction to pass the displayed value to -m pointed at an identifier the API does not accept. Prefer `canonical_name`, join `languages` for the display, and keep the older singular keys as the fallback for both. Regression tests cover the SDK-shaped entry in the default table and in the serialized JSON, plus the legacy shape. Co-Authored-By: Claude Opus 5 --- .../src/deepctl_cmd_speak/command.py | 10 +- .../tests/unit/test_speak_command.py | 102 ++++++++++++++++++ 2 files changed, 110 insertions(+), 2 deletions(-) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 79a81a5..3728039 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -906,12 +906,18 @@ def _list_voices(self, client: DeepgramClient) -> BaseResult: voices: list[VoiceInfo] = [] for m in result.get("tts", []): - name = m.get("name", "") + # canonical_name is the value -m takes ("aura-2-agathe-fr"); the + # bare "name" is just the voice ("agathe") and is not a usable + # model id. Languages moved to a list in the same catalog reshape, + # so keep the singular key as the fallback for both. + name = m.get("canonical_name") or m.get("name") or "" + languages = m.get("languages") or [] + language = ", ".join(languages) or m.get("language") or "" voices.append( VoiceInfo( name=name, voice_type=_voice_type_badge(name), - language=m.get("language", ""), + language=language, ) ) diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index 018f6f1..fc701bc 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -1,5 +1,6 @@ """Tests for speak command.""" +import json import struct import sys import wave @@ -978,6 +979,107 @@ def test_handle_list_voices_json_prints_no_table( captured = capsys.readouterr() assert captured.out == "" + @patch("deepctl_cmd_speak.command.get_output_format", return_value="default") + def test_handle_list_voices_sdk_shaped_catalog( + self, _fmt, command, mock_config, mock_auth_manager, mock_client, capsys + ): + """The table shows the canonical -m value and every language it lists. + + The catalog carries the usable model id as ``canonical_name`` and the + languages as a list; reading only ``name``/``language`` would print + "agathe" with a blank language, which -m cannot take. + """ + mock_client.list_models.return_value = { + "tts": [ + { + "name": "agathe", + "canonical_name": "aura-2-agathe-fr", + "languages": ["fr", "fr-FR"], + } + ] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.count == 1 + voice = result.voices[0] + assert voice.name == "aura-2-agathe-fr" + assert voice.voice_type == "aura" + assert voice.language == "fr, fr-FR" + + captured = capsys.readouterr() + assert "aura-2-agathe-fr" in captured.out + assert "fr, fr-FR" in captured.out + + @patch("deepctl_core.output.get_output_format", return_value="json") + @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") + def test_handle_list_voices_json_carries_canonical_name( + self, + _cmd_fmt, + _core_fmt, + command, + mock_config, + mock_auth_manager, + mock_client, + capsys, + ): + """The serialized JSON carries the same canonical id and languages.""" + mock_client.list_models.return_value = { + "tts": [ + { + "name": "agathe", + "canonical_name": "aura-2-agathe-fr", + "languages": ["fr", "fr-FR"], + } + ] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + command.output_result(result, mock_config) + + payload = json.loads(capsys.readouterr().out) + assert payload["voices"] == [ + { + "name": "aura-2-agathe-fr", + "voice_type": "aura", + "language": "fr, fr-FR", + } + ] + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="default") + def test_handle_list_voices_legacy_catalog_fields( + self, _fmt, command, mock_config, mock_auth_manager, mock_client + ): + """A catalog without the new keys still falls back to name/language.""" + mock_client.list_models.return_value = { + "tts": [{"name": "aura-2-asteria-en", "language": "en"}] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.voices[0].name == "aura-2-asteria-en" + assert result.voices[0].language == "en" + def test_handle_list_voices_empty( self, command, mock_config, mock_auth_manager, mock_client ): From a36a507599c4b8396d4975cffac1645480f0140d Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Wed, 23 Sep 2026 09:07:19 -0400 Subject: [PATCH 6/7] fix(speak): carry voice languages as data and flag the unlisted default Two follow-ups on --list-voices. The catalog endpoint returns no Flux voices, so the default the footer names (flux-alexis-en) is absent from a table that --model's help sends people to. The footer now says "(not listed above)" when the default is missing from the listing, computed rather than hardcoded so the note disappears once the catalog carries it. `-o json` handed consumers "fr, fr-FR", one display string they had to split. VoiceInfo gained a `languages` list alongside the joined `language` the table renders, so the published agent-facing command reference gives the tags as data. Co-Authored-By: Claude Opus 5 --- .../src/deepctl_cmd_speak/command.py | 23 +++++++++++++++---- .../src/deepctl_cmd_speak/models.py | 3 ++- .../tests/unit/test_speak_command.py | 10 ++++++++ 3 files changed, 31 insertions(+), 5 deletions(-) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 3728039..39bd3cd 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -911,13 +911,19 @@ def _list_voices(self, client: DeepgramClient) -> BaseResult: # model id. Languages moved to a list in the same catalog reshape, # so keep the singular key as the fallback for both. name = m.get("canonical_name") or m.get("name") or "" - languages = m.get("languages") or [] - language = ", ".join(languages) or m.get("language") or "" + languages = [str(lang) for lang in (m.get("languages") or [])] + legacy_language = m.get("language") or "" + if not languages and legacy_language: + languages = [legacy_language] voices.append( VoiceInfo( name=name, voice_type=_voice_type_badge(name), - language=language, + # The table wants one cell; -o json consumers want the + # tags as data, so carry both rather than make them split + # a display string. + language=", ".join(languages), + languages=languages, ) ) @@ -942,10 +948,19 @@ def _list_voices(self, client: DeepgramClient) -> BaseResult: table.add_row(v.name, v.voice_type, v.language) stdout_console.print(table) + # The catalog endpoint returns no Flux voices today, so the + # default is missing from the list --model's help points at. + # Say so rather than let it read as a typo; computed, so the + # note disappears once the catalog carries it. + default_note = ( + "" + if any(v.name == _DEFAULT_MODEL for v in voices) + else " (not listed above)" + ) stdout_console.print( f"\n[dim]{len(voices)} voice(s) — generate with " f'dg speak "..." -m ; ' - f"default: {_DEFAULT_MODEL}[/dim]" + f"default: {_DEFAULT_MODEL}{default_note}[/dim]" ) return SpeakVoicesResult( diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py index 177be57..45dcad2 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py @@ -16,7 +16,8 @@ class SpeakResult(BaseResult): class VoiceInfo(BaseModel): name: str = "" voice_type: str = "" # "aura" or "flux" - language: str = "" + language: str = "" # joined display string, e.g. "fr, fr-FR" + languages: list[str] = Field(default_factory=list) class SpeakVoicesResult(BaseResult): diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index fc701bc..7260cda 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -955,7 +955,9 @@ def test_handle_list_voices( captured = capsys.readouterr() assert "Deepgram TTS Voices" in captured.out assert "flux-alexis-en" in captured.out + # flux-alexis-en is one of the rows here, so the footer names it plainly. assert "default: flux-alexis-en" in " ".join(captured.out.split()) + assert "not listed above" not in captured.out @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") def test_handle_list_voices_json_prints_no_table( @@ -1013,10 +1015,16 @@ def test_handle_list_voices_sdk_shaped_catalog( assert voice.name == "aura-2-agathe-fr" assert voice.voice_type == "aura" assert voice.language == "fr, fr-FR" + assert voice.languages == ["fr", "fr-FR"] captured = capsys.readouterr() assert "aura-2-agathe-fr" in captured.out assert "fr, fr-FR" in captured.out + # This catalog carries no Flux voices, so the footer's default is not + # one of the rows above it and the table says so. + assert "default: flux-alexis-en (not listed above)" in " ".join( + captured.out.split() + ) @patch("deepctl_core.output.get_output_format", return_value="json") @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") @@ -1056,6 +1064,7 @@ def test_handle_list_voices_json_carries_canonical_name( "name": "aura-2-agathe-fr", "voice_type": "aura", "language": "fr, fr-FR", + "languages": ["fr", "fr-FR"], } ] @@ -1079,6 +1088,7 @@ def test_handle_list_voices_legacy_catalog_fields( assert isinstance(result, SpeakVoicesResult) assert result.voices[0].name == "aura-2-asteria-en" assert result.voices[0].language == "en" + assert result.voices[0].languages == ["en"] def test_handle_list_voices_empty( self, command, mock_config, mock_auth_manager, mock_client From edf9d329214ae942a4826c9fdae74993592e761e Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Thu, 1 Oct 2026 12:07:37 -0400 Subject: [PATCH 7/7] fix(speak): gate --play per player instead of treating paplay like aplay Review finding S1: paplay sat in a PCM/WAV-only list alongside aplay, so `--play` refused Aura FLAC and Opus output before the API call even though paplay decodes both through libsndfile. The preflight now consults a per-player table of encodings a player cannot decode: aplay rejects every compressed encoding, paplay rejects only mp3 and aac, and ffplay and afplay have no entry (afplay decodes Ogg Opus on current macOS; checked with a generated probe file). The error names what the player can play and the --encoding way out. Also adopts main's post-#107 convention: the command's stderr console is get_status_console() rather than a private Console(stderr=True), so the agentic no-color settings apply to --play status lines. README paragraph updated to match; regression tests cover each player's accept and reject set. Co-Authored-By: Claude Fable 5.1 --- README.md | 8 ++-- .../src/deepctl_cmd_speak/command.py | 30 +++++++++---- .../tests/unit/test_speak_command.py | 42 ++++++++++++++++--- 3 files changed, 64 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index ebe14f9..abf6873 100644 --- a/README.md +++ b/README.md @@ -214,9 +214,11 @@ dg speak "Hola, bienvenido a Deepgram" -o hola.mp3 -m aura-2-selena-es `--play` uses the first available system player (`ffplay`, `afplay`, `paplay`, or `aplay`); install `ffmpeg` if none is present. With the Flux default the audio is streamed into the player as it arrives, so playback starts at -first-audio latency rather than after the whole utterance. `paplay` and `aplay` -decode PCM/WAV only, so playing Aura's MP3 output needs `ffplay` (or `afplay` -on macOS). +first-audio latency rather than after the whole utterance. Each fallback +player is checked against the requested format before the API call: `aplay` +plays PCM/WAV only and `paplay` adds FLAC and Opus but not MP3 or AAC, while +`ffplay` and `afplay` (macOS) play every format. Playing Aura's default MP3 on +Linux needs `ffplay` or `--encoding flac`. ### Text Intelligence diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 39bd3cd..78738ef 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -22,6 +22,7 @@ Config, DeepgramClient, get_output_format, + get_status_console, ) from rich.console import Console from rich.markup import escape @@ -32,7 +33,7 @@ if TYPE_CHECKING: from collections.abc import Iterator -console = Console(stderr=True) +console = get_status_console() # Tables and other stdout-bound rendering (only used when no audio goes to # stdout, i.e. --list-voices). stdout_console = Console() @@ -63,10 +64,6 @@ # Players with no stdin mode, which are handed a temp file by path instead. _FILE_PLAYER_ARGV = {"afplay": ["afplay"]} -# paplay and aplay decode PCM/WAV only -- they cannot play a compressed -# container such as Aura's default mp3. -_PCM_ONLY_PLAYERS = ("paplay", "aplay") - # Compressed, self-describing encodings: a player can sniff these from the # byte stream. The PCM encodings cannot be sniffed — Speak v1 wraps them in a # WAV container unless `--container none` is passed, and Flux wraps its @@ -74,6 +71,16 @@ _COMPRESSED_ENCODINGS = ("mp3", "aac", "opus", "flac") _PCM_ENCODINGS = ("linear16", "mulaw", "alaw") +# Compressed encodings each fallback player cannot decode, checked before the +# API call. ffplay (ffmpeg) and afplay (CoreAudio) play everything deepctl +# emits, so they have no entry. paplay decodes through libsndfile, which reads +# WAV, FLAC, and Ogg (Vorbis and Opus) but not mp3 or aac on the distro builds +# in common use. aplay plays PCM/WAV only. +_UNPLAYABLE_BY_PLAYER: dict[str, tuple[str, ...]] = { + "paplay": ("mp3", "aac"), + "aplay": _COMPRESSED_ENCODINGS, +} + # Encoding/container -> temp-file suffix, so the temp file handed to afplay is # sniffed correctly by CoreAudio. _SUFFIX_BY_ENCODING = { @@ -145,10 +152,17 @@ def _check_playable( "--container wav (Aura), or save it with -o." ) - if compressed and player in _PCM_ONLY_PLAYERS: + unplayable = _UNPLAYABLE_BY_PLAYER.get(player, ()) + if compressed and eff_encoding in unplayable: + playable = ", ".join( + "linear16 (WAV)" if e == "linear16" else e + for e in ("linear16", *_COMPRESSED_ENCODINGS) + if e not in unplayable + ) return ( - f"'{player}' can only play PCM/WAV audio, not {eff_encoding}. " - "Install ffmpeg (ffplay) to play it, or save it with -o." + f"'{player}' cannot decode {eff_encoding}; it plays {playable}. " + "Pick one of those with --encoding, install ffmpeg (ffplay) to play " + f"{eff_encoding}, or save it with -o." ) return None diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index 7260cda..df476f8 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -1730,7 +1730,7 @@ def test_handle_play_rejects_mp3_on_pcm_only_player( assert result is not None assert result.status == "error" - assert "can only play PCM/WAV" in result.message + assert "cannot decode mp3" in result.message mock_client.speak_text.assert_not_called() @@ -1817,13 +1817,45 @@ def test_check_playable_rejects_raw_audio(self): assert aura_raw is not None assert "no container" in aura_raw - def test_check_playable_rejects_compressed_audio_on_pcm_only_players(self): - for player in ("paplay", "aplay"): + def test_check_playable_aplay_rejects_every_compressed_encoding(self): + for encoding in ("mp3", "aac", "opus", "flac"): problem = _check_playable( - player, is_flux=False, encoding="mp3", container=None + "aplay", is_flux=False, encoding=encoding, container=None ) assert problem is not None - assert "can only play PCM/WAV" in problem + assert f"cannot decode {encoding}" in problem + assert "linear16 (WAV)" in problem + + def test_check_playable_paplay_decodes_flac_and_ogg_but_not_mp3_or_aac(self): + """paplay reads FLAC and Ogg through libsndfile; only mp3/aac are out. + + Regression for the review finding that paplay was gated like aplay and + refused FLAC and Opus it can play. + """ + for encoding in ("flac", "opus"): + assert ( + _check_playable( + "paplay", is_flux=False, encoding=encoding, container=None + ) + is None + ) + for encoding in ("mp3", "aac"): + problem = _check_playable( + "paplay", is_flux=False, encoding=encoding, container=None + ) + assert problem is not None + assert f"'paplay' cannot decode {encoding}" in problem + assert "flac" in problem and "opus" in problem + + def test_check_playable_ffplay_and_afplay_accept_every_compressed_encoding(self): + for player in ("ffplay", "afplay"): + for encoding in ("mp3", "aac", "opus", "flac"): + assert ( + _check_playable( + player, is_flux=False, encoding=encoding, container=None + ) + is None + ) def test_play_audio_pipes_to_stdin_player(self, stub_stdin_player): """The bytes handed to a stdin player are exactly the audio."""