diff --git a/README.md b/README.md index 3d30eac..abf6873 100644 --- a/README.md +++ b/README.md @@ -184,8 +184,15 @@ container so it is directly playable. Pass an `aura-*` model to use the Speak v1 batch REST API instead, which supports containerized formats like MP3. ```bash +# Quickstart — synthesize and hear it straight away +dg speak "Hello from Deepgram" --play + +# Discover voices +dg speak --list-voices + # Flux TTS (v2, WebSocket streaming) — the default dg speak "Hello from Flux" -o hello.wav +dg speak "Save it and play it" -o hello.wav --play # Piped audio is a streaming WAV; -loglevel error hides ffmpeg's cosmetic # end-of-stream notice (the audio is complete). dg speak "Hello from Flux" | ffplay -loglevel error -nodisp -autoexit - @@ -204,6 +211,15 @@ echo "Hello" | dg speak -o greeting.mp3 -m aura-2-asteria-en dg speak "Hola, bienvenido a Deepgram" -o hola.mp3 -m aura-2-selena-es ``` +`--play` uses the first available system player (`ffplay`, `afplay`, `paplay`, +or `aplay`); install `ffmpeg` if none is present. With the Flux default the +audio is streamed into the player as it arrives, so playback starts at +first-audio latency rather than after the whole utterance. Each fallback +player is checked against the requested format before the API call: `aplay` +plays PCM/WAV only and `paplay` adds FLAC and Opus but not MP3 or AAC, while +`ffplay` and `afplay` (macOS) play every format. Playing Aura's default MP3 on +Linux needs `ffplay` or `--encoding flac`. + ### Text Intelligence Analyze text for sentiment, summaries, topics, and intents. diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py index 705a32b..78738ef 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/command.py @@ -3,12 +3,16 @@ from __future__ import annotations import io +import shutil import struct +import subprocess import sys +import tempfile import time import wave +from contextlib import contextmanager, suppress from pathlib import Path -from typing import TYPE_CHECKING, Any +from typing import IO, TYPE_CHECKING, Any import click from deepctl_core import ( @@ -17,15 +21,26 @@ BaseResult, Config, DeepgramClient, + get_output_format, + get_status_console, ) from rich.console import Console +from rich.markup import escape +from rich.table import Table -from .models import SpeakResult +from .models import SpeakResult, SpeakVoicesResult, VoiceInfo if TYPE_CHECKING: from collections.abc import Iterator -console = Console(stderr=True) +console = get_status_console() +# Tables and other stdout-bound rendering (only used when no audio goes to +# stdout, i.e. --list-voices). +stdout_console = Console() + +# Keep in sync with the --model option default below; also reported by +# --list-voices so the table says which voice you get for free. +_DEFAULT_MODEL = "flux-alexis-en" # Flux (Speak v2) streaming controls, per the /v2/speak API. `speed` is a # 0.05-increment multiplier and `expressivity` is a small integer range; both @@ -34,6 +49,195 @@ _FLUX_SPEEDS = (0.85, 0.90, 0.95, 1.00, 1.05, 1.10, 1.15) _FLUX_EXPRESSIVITY = (-2, -1, 0, 1, 2) +# Audio players probed for --play, in preference order. ffplay ships with +# ffmpeg and plays every format we can emit; the rest are OS-native fallbacks. +_AUDIO_PLAYERS = ("ffplay", "afplay", "paplay", "aplay") + +# Players that read audio from stdin, and the argv that makes them do it. +# afplay (macOS) has no stdin mode, so it is handed a temp file instead. +_STDIN_PLAYER_ARGV = { + "ffplay": ["ffplay", "-loglevel", "error", "-nodisp", "-autoexit", "-"], + "paplay": ["paplay"], + "aplay": ["aplay", "-q", "-"], +} + +# Players with no stdin mode, which are handed a temp file by path instead. +_FILE_PLAYER_ARGV = {"afplay": ["afplay"]} + +# Compressed, self-describing encodings: a player can sniff these from the +# byte stream. The PCM encodings cannot be sniffed — Speak v1 wraps them in a +# WAV container unless `--container none` is passed, and Flux wraps its +# linear16 stream itself. +_COMPRESSED_ENCODINGS = ("mp3", "aac", "opus", "flac") +_PCM_ENCODINGS = ("linear16", "mulaw", "alaw") + +# Compressed encodings each fallback player cannot decode, checked before the +# API call. ffplay (ffmpeg) and afplay (CoreAudio) play everything deepctl +# emits, so they have no entry. paplay decodes through libsndfile, which reads +# WAV, FLAC, and Ogg (Vorbis and Opus) but not mp3 or aac on the distro builds +# in common use. aplay plays PCM/WAV only. +_UNPLAYABLE_BY_PLAYER: dict[str, tuple[str, ...]] = { + "paplay": ("mp3", "aac"), + "aplay": _COMPRESSED_ENCODINGS, +} + +# Encoding/container -> temp-file suffix, so the temp file handed to afplay is +# sniffed correctly by CoreAudio. +_SUFFIX_BY_ENCODING = { + "mp3": ".mp3", + "linear16": ".wav", + "flac": ".flac", + "opus": ".ogg", + "aac": ".aac", +} + + +def _find_audio_player() -> str | None: + """Return the first available audio player command, or ``None``.""" + for player in _AUDIO_PLAYERS: + if shutil.which(player): + return player + return None + + +def _voice_type_badge(model_name: str) -> str: + """Derive an aura/flux badge from a TTS model name prefix.""" + name = model_name.lower() + if name.startswith("aura"): + return "aura" + if name.startswith("flux"): + return "flux" + return "tts" + + +def _play_suffix(*, is_flux: bool, encoding: str | None, container: str | None) -> str: + """Best-effort file suffix for the audio handed to afplay's temp file.""" + if is_flux: + # Flux streams raw audio; only linear16 gets a WAV wrapper (below). + return ".wav" if (encoding or "linear16") == "linear16" else ".raw" + if container == "wav": + return ".wav" + if container == "ogg": + return ".ogg" + eff_encoding = (encoding or "mp3").lower() + if eff_encoding in _PCM_ENCODINGS: + # Speak v1 defaults these to a WAV container; `--container none` is + # the only way to get them bare. + return ".raw" if container == "none" else ".wav" + return _SUFFIX_BY_ENCODING.get(eff_encoding, ".mp3") + + +def _check_playable( + player: str, *, is_flux: bool, encoding: str | None, container: str | None +) -> str | None: + """Return why ``player`` cannot play this audio format, or ``None``. + + Checked before the API call so a format the chosen player cannot decode + fails with an explanation instead of silence or a decoder error. + """ + eff_encoding = (encoding or ("linear16" if is_flux else "mp3")).lower() + compressed = eff_encoding in _COMPRESSED_ENCODINGS + + # Raw PCM with no container is undetectable: the player has no way to know + # the sample rate, width, or encoding. Flux linear16 gets a WAV wrapper + # (so it is fine); Flux mulaw/alaw and an explicit `--container none` do + # not. + raw = (is_flux and eff_encoding != "linear16") or ( + not is_flux and not compressed and container == "none" + ) + if raw: + return ( + f"--play cannot play raw {eff_encoding} audio: it has no container " + "for the player to detect. Use the default linear16 audio, add " + "--container wav (Aura), or save it with -o." + ) + + unplayable = _UNPLAYABLE_BY_PLAYER.get(player, ()) + if compressed and eff_encoding in unplayable: + playable = ", ".join( + "linear16 (WAV)" if e == "linear16" else e + for e in ("linear16", *_COMPRESSED_ENCODINGS) + if e not in unplayable + ) + return ( + f"'{player}' cannot decode {eff_encoding}; it plays {playable}. " + "Pick one of those with --encoding, install ffmpeg (ffplay) to play " + f"{eff_encoding}, or save it with -o." + ) + return None + + +@contextmanager +def _player_stdin(player: str) -> Iterator[IO[bytes]]: + """Spawn a stdin-reading player and yield its stdin for streaming writes. + + On a clean exit the pipe is closed (which is how the player learns the + audio ended) and the process is waited for, so the command does not return + before playback finishes. If the body raises, the player is killed rather + than left holding a half-written stream. + """ + proc = subprocess.Popen(_STDIN_PLAYER_ARGV[player], stdin=subprocess.PIPE) + stdin = proc.stdin + assert stdin is not None # stdin=PIPE above + try: + yield stdin + except BaseException: + proc.kill() + proc.wait() + raise + finally: + # A player that quit early (ffplay's "q") leaves a broken pipe; that + # is the user stopping playback, not a failure. + with suppress(BrokenPipeError, OSError): + stdin.close() + returncode = proc.wait() + if returncode: + raise click.ClickException( + f"Audio player '{player}' exited with status {returncode}." + ) + + +def _play_audio(player: str, audio: bytes, *, suffix: str) -> None: + """Play a fully-synthesized blob through the detected system player. + + ffplay/paplay/aplay read it from stdin; afplay (macOS) has no stdin mode, + so the bytes go to a temp file that is played by path. Playback failures + raise so the command exits non-zero rather than reporting a success it + did not deliver. + """ + if player in _FILE_PLAYER_ARGV: + with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as tmp: + tmp.write(audio) + tmp_path = tmp.name + try: + returncode = subprocess.run( + [*_FILE_PLAYER_ARGV[player], tmp_path], check=False + ).returncode + finally: + Path(tmp_path).unlink(missing_ok=True) + if returncode: + raise click.ClickException( + f"Audio player '{player}' exited with status {returncode}." + ) + return + + with _player_stdin(player) as sink, suppress(BrokenPipeError): + sink.write(audio) + + +def _fail(message: str) -> BaseResult: + """An error result whose message the user actually sees. + + In default output mode the framework prints nothing for a returned + result — it only maps the status to a non-zero exit code — so the human + message has to go to the stderr console here, the way every other status + line in this command does. The message is escaped because it can carry a + user-supplied path, and rich would otherwise eat any [bracketed] segment + of it as markup. + """ + console.print(f"[red]Error:[/red] {escape(message)}") + return BaseResult(status="error", message=message) + def _fmt_bytes(n: int) -> str: """Human-readable byte count for progress display.""" @@ -173,8 +377,11 @@ class SpeakCommand(BaseCommand): # raw audio wrapped in WAV. Piped audio is a streaming WAV (unknown # length up front), so pass `-loglevel error` to silence ffmpeg's # cosmetic end-of-stream notice. + 'dg speak "Hello from Deepgram" --play', + "dg speak --list-voices", 'dg speak "Hello world"', 'dg speak "Hello world" -o hello.wav', + 'dg speak "Hello world" -o hello.wav --play', "dg speak --file message.txt -o output.wav", 'dg speak "Hello" | ffplay -loglevel error -nodisp -autoexit -', # Flux TTS streaming controls: --speed (0.85-1.15) and beta @@ -200,7 +407,9 @@ class SpeakCommand(BaseCommand): "formats like mp3. Supports model selection and audio format options. " "Flux TTS models also accept --speed (0.85–1.15) and beta " "--expressivity (-2..2; default 0 = nominal) streaming controls; these " - "are rejected for other models." + "are rejected for other models. --play sends the audio to a local " + "system player instead of (or as well as) a file, and --list-voices " + "prints the available TTS voices without generating speech." ) def get_arguments(self) -> list[dict[str, Any]]: @@ -222,11 +431,12 @@ def get_arguments(self) -> list[dict[str, Any]]: "help": ( "TTS model. flux-* = Flux TTS / Speak v2 (WebSocket " "streaming; default flux-alexis-en); aura-* = Speak v1 " - "(REST batch, e.g. aura-2-asteria-en)." + "(REST batch, e.g. aura-2-asteria-en). " + "See voices: dg speak --list-voices" ), "type": str, "is_option": True, - "default": "flux-alexis-en", + "default": _DEFAULT_MODEL, }, { "names": ["--encoding"], @@ -278,6 +488,21 @@ def get_arguments(self) -> list[dict[str, Any]]: "type": str, "is_option": True, }, + { + "names": ["--play"], + "help": ( + "Play the audio through your system player (ffplay, afplay, " + "paplay, or aplay). Can be combined with -o to save and play." + ), + "is_flag": True, + "is_option": True, + }, + { + "names": ["--list-voices"], + "help": "List available TTS voices and exit (no speech generated)", + "is_flag": True, + "is_option": True, + }, ] def handle( @@ -289,43 +514,68 @@ def handle( ) -> BaseResult | None: text = kwargs.get("text") output_path = kwargs.get("output") - model = kwargs.get("model") or "flux-alexis-en" + model = kwargs.get("model") or _DEFAULT_MODEL encoding = kwargs.get("encoding") container = kwargs.get("container") sample_rate = kwargs.get("sample_rate") speed = kwargs.get("speed") expressivity = kwargs.get("expressivity") file_path = kwargs.get("file") + play = kwargs.get("play", False) + list_voices = kwargs.get("list_voices", False) + + # --list-voices: discover voices and exit before any TTS generation. + if list_voices: + return self._list_voices(client) # Resolve text input: arg > --file > stdin if not text and file_path: path = Path(file_path) if not path.exists(): - return BaseResult( - status="error", message=f"File not found: {file_path}" - ) + return _fail(f"File not found: {file_path}") text = path.read_text().strip() elif not text and not sys.stdin.isatty(): text = sys.stdin.read().strip() if not text: - return BaseResult( - status="error", - message="No text provided. Pass text as argument, use --file, or pipe via stdin.", + return _fail( + "No text provided. Pass text as argument, use --file, " + "or pipe via stdin." ) - # If stdout is a TTY and no output file, require --output + # If --play was requested, resolve the player up front so we fail fast + # with a clear message before spending an API call. + player: str | None = None + if play: + player = _find_audio_player() + if player is None: + return _fail( + "No audio player found — install ffmpeg, or use -o to save a file." + ) + + # If stdout is a TTY and there's nowhere for the audio to go, require an + # explicit destination. --play satisfies that requirement. stdout_is_tty = sys.stdout.isatty() - if not output_path and stdout_is_tty: - return BaseResult( - status="error", - message="No output specified. Use -o/--output to save to file, or pipe stdout.", + if not output_path and not play and stdout_is_tty: + return _fail( + "No output specified. Use -o/--output to save to a file, " + "--play to hear it, or pipe stdout." ) # Only the documented flux-* namespace uses speak.v2. Aura and unknown # model names pass through to the REST API so the service can resolve them. is_flux = model.lower().startswith("flux-") + # A player that cannot decode what we are about to request should say + # so now, before the API call, rather than emit silence or a decoder + # error after synthesis. + if player is not None: + unplayable = _check_playable( + player, is_flux=is_flux, encoding=encoding, container=container + ) + if unplayable is not None: + return _fail(unplayable) + # speed / expressivity are Flux (Speak v2) connect controls; reject them # for other models rather than silently dropping them. Raise (not return) # so the failure exits non-zero in every output mode. @@ -346,6 +596,15 @@ def handle( f"--expressivity must be one of: {allowed} (got {expressivity})." ) + if play and not output_path and not stdout_is_tty: + # Playing sends the audio to the player, not to stdout, so a + # redirect or pipe would otherwise collect nothing and give no + # hint why. + console.print( + "[yellow]Note:[/yellow] --play sends the audio to your player, " + "so stdout stays empty. Add -o to save a file too." + ) + if is_flux: # WebSocket streaming path. Streaming output is raw audio, so # default to linear16 @ 24kHz and wrap it in WAV for playback. @@ -376,10 +635,97 @@ def handle( ) ) - if output_path: - # A WAV file must declare its data length in the header, which - # we only know once the stream ends — so buffer, then wrap and - # write. (Streaming to disk has no user-visible benefit here.) + # Flux linear16 into a stdin-reading player: stream it. The + # player gets the streaming WAV header before the first frame and + # then every frame as Flux emits it, so sound starts at + # first-audio latency instead of after the whole utterance — the + # same framing `dg speak | ffplay -` relies on. + if ( + player is not None + and player in _STDIN_PLAYER_ARGV + and eff_encoding == "linear16" + ): + console.print(f"[blue]Playing audio ({player})...[/blue]") + pcm = bytearray() + saved_bytes = 0 + player_gone = False + try: + with _player_stdin(player) as sink: + wrote_header = False + for chunk in stream: + if not chunk: + continue + if output_path: + pcm.extend(chunk) + if player_gone: + continue + try: + if not wrote_header: + sink.write( + _streaming_wav_header( + sample_rate=int(eff_sample_rate) + ) + ) + wrote_header = True + sink.write(chunk) + sink.flush() + except BrokenPipeError: + # The player exited first (ffplay's "q", for + # instance). That is the user stopping + # playback, not a failure — but keep draining + # the stream so a -o file is still complete. + player_gone = True + + if output_path and pcm: + # Save before waiting on the player (that wait + # happens on leaving this block), so stopping + # playback still leaves the file behind. The file + # gets a real, exact-length header rather than the + # streaming placeholder the player was handed. + audio_bytes = _pcm_to_wav( + bytes(pcm), sample_rate=int(eff_sample_rate) + ) + Path(output_path).write_bytes(audio_bytes) + saved_bytes = len(audio_bytes) + except click.ClickException: + raise + except Exception as e: + raise click.ClickException(f"Flux streaming failed: {e}") + + if prog.total == 0: + raise click.ClickException( + "Flux (Speak v2) streaming returned no audio." + ) + + if output_path: + console.print( + f"[green]Audio saved to {output_path}[/green] " + f"({saved_bytes:,} bytes — {prog.timing()})" + ) + else: + console.print( + f"[green]✓ Played {prog.total:,} bytes[/green] " + f"({prog.timing()})" + ) + return SpeakResult( + status="success", + message=( + f"Audio saved to {output_path}" + if output_path + else f"Played {prog.total:,} bytes" + ), + output_path=output_path or "", + model=model, + bytes_written=saved_bytes or prog.total, + played=True, + ) + + if output_path or player: + # Saving, or playing through afplay (which has no stdin mode), + # needs the full utterance: a WAV file must declare its data + # length in the header (known only once the stream ends), and + # afplay is handed a complete file. So buffer, then wrap once + # and reuse for both sinks. pcm = bytearray() try: for chunk in stream: @@ -403,17 +749,31 @@ def handle( audio_bytes = bytes(pcm) total_bytes = len(audio_bytes) - Path(output_path).write_bytes(audio_bytes) - console.print( - f"[green]Audio saved to {output_path}[/green] " - f"({total_bytes:,} bytes — {prog.timing()})" - ) + if output_path: + Path(output_path).write_bytes(audio_bytes) + console.print( + f"[green]Audio saved to {output_path}[/green] " + f"({total_bytes:,} bytes — {prog.timing()})" + ) + + if player: + suffix = _play_suffix( + is_flux=True, encoding=eff_encoding, container=None + ) + console.print(f"[blue]Playing audio ({player})...[/blue]") + _play_audio(player, audio_bytes, suffix=suffix) + return SpeakResult( status="success", - message=f"Audio saved to {output_path}", - output_path=output_path, + message=( + f"Audio saved to {output_path}" + if output_path + else f"Played {total_bytes:,} bytes" + ), + output_path=output_path or "", model=model, bytes_written=total_bytes, + played=bool(player), ) # Pipe path — write each chunk to stdout as it arrives so a @@ -469,7 +829,45 @@ def handle( total_bytes = 0 - if output_path: + if player: + # Playback needs the full utterance, so buffer it. When -o is + # also given, save the same bytes to the file too. + audio = bytearray() + for chunk in audio_iter: + audio.extend(chunk) + audio_bytes = bytes(audio) + total_bytes = len(audio_bytes) + + if not audio_bytes: + return _fail("TTS returned no audio.") + + if output_path: + Path(output_path).write_bytes(audio_bytes) + console.print( + f"[green]Audio saved to {output_path}[/green] " + f"({total_bytes:,} bytes)" + ) + + suffix = _play_suffix( + is_flux=False, encoding=encoding, container=container + ) + console.print(f"[blue]Playing audio ({player})...[/blue]") + _play_audio(player, audio_bytes, suffix=suffix) + + message = ( + f"Audio saved to {output_path}" + if output_path + else f"Played {total_bytes:,} bytes" + ) + return SpeakResult( + status="success", + message=message, + output_path=output_path or "", + model=model, + bytes_written=total_bytes, + played=True, + ) + elif output_path: # Write to file out = Path(output_path) with open(out, "wb") as f: @@ -511,3 +909,76 @@ def handle( # still exits 0. Reachable now that unknown models (e.g. a bare # "flux" typo) route here instead of the raising v2 path. raise click.ClickException(f"Error generating speech: {e}") + + def _list_voices(self, client: DeepgramClient) -> BaseResult: + """List available TTS voices as a table (mirrors `dg models`).""" + try: + result = client.list_models() + except Exception as e: + console.print(f"[red]Error listing voices:[/red] {e}") + return BaseResult(status="error", message=str(e)) + + voices: list[VoiceInfo] = [] + for m in result.get("tts", []): + # canonical_name is the value -m takes ("aura-2-agathe-fr"); the + # bare "name" is just the voice ("agathe") and is not a usable + # model id. Languages moved to a list in the same catalog reshape, + # so keep the singular key as the fallback for both. + name = m.get("canonical_name") or m.get("name") or "" + languages = [str(lang) for lang in (m.get("languages") or [])] + legacy_language = m.get("language") or "" + if not languages and legacy_language: + languages = [legacy_language] + voices.append( + VoiceInfo( + name=name, + voice_type=_voice_type_badge(name), + # The table wants one cell; -o json consumers want the + # tags as data, so carry both rather than make them split + # a display string. + language=", ".join(languages), + languages=languages, + ) + ) + + if not voices: + console.print("[yellow]No TTS voices found[/yellow]") + return SpeakVoicesResult(status="info", message="No voices found") + + # Render the human table only in default mode: for json/yaml/csv the + # framework serializes the returned result to stdout, so a table here + # would corrupt what callers pipe into jq. (No audio goes to stdout on + # this path, so the table is safe to print there, same as `dg models`.) + if get_output_format() == "default": + table = Table( + title="Deepgram TTS Voices", + show_header=True, + header_style="bold blue", + ) + table.add_column("Voice", style="green") + table.add_column("Type", style="cyan") + table.add_column("Language") + for v in voices: + table.add_row(v.name, v.voice_type, v.language) + + stdout_console.print(table) + # The catalog endpoint returns no Flux voices today, so the + # default is missing from the list --model's help points at. + # Say so rather than let it read as a typo; computed, so the + # note disappears once the catalog carries it. + default_note = ( + "" + if any(v.name == _DEFAULT_MODEL for v in voices) + else " (not listed above)" + ) + stdout_console.print( + f"\n[dim]{len(voices)} voice(s) — generate with " + f'dg speak "..." -m ; ' + f"default: {_DEFAULT_MODEL}{default_note}[/dim]" + ) + + return SpeakVoicesResult( + status="success", + voices=voices, + count=len(voices), + ) diff --git a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py index 8e483db..45dcad2 100644 --- a/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py +++ b/packages/deepctl-cmd-speak/src/deepctl_cmd_speak/models.py @@ -3,9 +3,23 @@ from __future__ import annotations from deepctl_core import BaseResult +from pydantic import BaseModel, Field class SpeakResult(BaseResult): output_path: str = "" model: str = "" bytes_written: int = 0 + played: bool = False + + +class VoiceInfo(BaseModel): + name: str = "" + voice_type: str = "" # "aura" or "flux" + language: str = "" # joined display string, e.g. "fr, fr-FR" + languages: list[str] = Field(default_factory=list) + + +class SpeakVoicesResult(BaseResult): + voices: list[VoiceInfo] = Field(default_factory=list) + count: int = 0 diff --git a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py index f37952b..df476f8 100644 --- a/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py +++ b/packages/deepctl-cmd-speak/tests/unit/test_speak_command.py @@ -1,18 +1,126 @@ """Tests for speak command.""" +import json +import struct +import sys +import wave +from contextlib import contextmanager +from pathlib import Path from unittest.mock import Mock, patch import click import pytest from deepctl_cmd_speak.command import ( + _FILE_PLAYER_ARGV, + _STDIN_PLAYER_ARGV, SpeakCommand, + _check_playable, + _find_audio_player, _fmt_bytes, + _play_audio, + _play_suffix, + _player_stdin, + _streaming_wav_header, _StreamProgress, + _voice_type_badge, ) -from deepctl_cmd_speak.models import SpeakResult +from deepctl_cmd_speak.models import SpeakResult, SpeakVoicesResult from deepctl_core import AuthManager, BaseResult, Config, DeepgramClient +def _only(*players): + """A shutil.which side effect where only these players are installed.""" + return lambda p: f"/usr/bin/{p}" if p in players else None + + +def _wav_header_fields(blob: bytes) -> tuple[int, int, int]: + """(sample_rate, channels, bits) parsed from a 44-byte PCM WAV header.""" + assert blob[:4] == b"RIFF" + assert blob[8:12] == b"WAVE" + assert blob[12:16] == b"fmt " + channels, sample_rate = struct.unpack(" None: + self.copy_path = copy_path + self._record_path = record_path + + @property + def arg_path(self) -> str: + return self._record_path.read_text().strip() + + +@pytest.fixture +def stub_file_player(tmp_path, monkeypatch): + """Stand a real stub process in for afplay, which plays a file by path.""" + script = tmp_path / "file_player.py" + script.write_text( + "import sys\n" + "src = sys.argv[3]\n" + "with open(src, 'rb') as fh:\n" + " data = fh.read()\n" + "with open(sys.argv[1], 'wb') as fh:\n" + " fh.write(data)\n" + "with open(sys.argv[2], 'w') as fh:\n" + " fh.write(src)\n" + ) + copy_path = tmp_path / "played.bin" + record_path = tmp_path / "played_path.txt" + monkeypatch.setitem( + _FILE_PLAYER_ARGV, + "afplay", + [sys.executable, str(script), str(copy_path), str(record_path)], + ) + return _FilePlayerStub(copy_path, record_path) + + +@pytest.fixture +def failing_file_player(tmp_path, monkeypatch): + """A file-based player that exits non-zero.""" + script = tmp_path / "failing_file_player.py" + script.write_text("import sys\nsys.exit(4)\n") + monkeypatch.setitem(_FILE_PLAYER_ARGV, "afplay", [sys.executable, str(script)]) + return script + + class TestStreamProgress: """The live-progress helper used by the Flux streaming path.""" @@ -104,6 +212,8 @@ def test_get_arguments(self, command): assert "--expressivity" in option_names assert "--file" in option_names assert "-f" in option_names + assert "--play" in option_names + assert "--list-voices" in option_names @patch("deepctl_cmd_speak.command.sys") def test_handle_no_text_error( @@ -238,6 +348,8 @@ def test_handle_no_output_tty_error( assert isinstance(result, BaseResult) assert result.status == "error" assert "No output specified" in result.message + # The message names every way out, including --play. + assert "--play" in result.message @patch("deepctl_cmd_speak.command.sys") def test_handle_write_to_file( @@ -797,6 +909,1002 @@ def test_handle_write_to_stdout( mock_stdout_buffer.write.assert_any_call(b"chunk2") mock_stdout_buffer.flush.assert_called_once() + # ── --list-voices ──────────────────────────────────────────────── + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="default") + def test_handle_list_voices( + self, _fmt, command, mock_config, mock_auth_manager, mock_client, capsys + ): + """--list-voices lists TTS voices with aura/flux badges, no generation.""" + mock_client.list_models.return_value = { + "stt": [{"name": "nova-3", "language": "en"}], + "tts": [ + {"name": "aura-2-asteria-en", "language": "en"}, + {"name": "flux-alexis-en", "language": "en"}, + ], + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + output=None, + model=None, + encoding=None, + container=None, + sample_rate=None, + file=None, + play=False, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.status == "success" + assert result.count == 2 + badges = {v.name: v.voice_type for v in result.voices} + assert badges["aura-2-asteria-en"] == "aura" + assert badges["flux-alexis-en"] == "flux" + # STT models are excluded; no speech is generated. + assert all("nova" not in v.name for v in result.voices) + mock_client.speak_text.assert_not_called() + mock_client.speak_text_stream.assert_not_called() + + # The table renders like `dg models`: on stdout, which carries no + # audio on this path, naming both voices and the default model. + captured = capsys.readouterr() + assert "Deepgram TTS Voices" in captured.out + assert "flux-alexis-en" in captured.out + # flux-alexis-en is one of the rows here, so the footer names it plainly. + assert "default: flux-alexis-en" in " ".join(captured.out.split()) + assert "not listed above" not in captured.out + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") + def test_handle_list_voices_json_prints_no_table( + self, _fmt, command, mock_config, mock_auth_manager, mock_client, capsys + ): + """In json mode the framework serializes the result; no table on stdout.""" + mock_client.list_models.return_value = { + "tts": [{"name": "flux-alexis-en", "language": "en"}] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.count == 1 + captured = capsys.readouterr() + assert captured.out == "" + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="default") + def test_handle_list_voices_sdk_shaped_catalog( + self, _fmt, command, mock_config, mock_auth_manager, mock_client, capsys + ): + """The table shows the canonical -m value and every language it lists. + + The catalog carries the usable model id as ``canonical_name`` and the + languages as a list; reading only ``name``/``language`` would print + "agathe" with a blank language, which -m cannot take. + """ + mock_client.list_models.return_value = { + "tts": [ + { + "name": "agathe", + "canonical_name": "aura-2-agathe-fr", + "languages": ["fr", "fr-FR"], + } + ] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.count == 1 + voice = result.voices[0] + assert voice.name == "aura-2-agathe-fr" + assert voice.voice_type == "aura" + assert voice.language == "fr, fr-FR" + assert voice.languages == ["fr", "fr-FR"] + + captured = capsys.readouterr() + assert "aura-2-agathe-fr" in captured.out + assert "fr, fr-FR" in captured.out + # This catalog carries no Flux voices, so the footer's default is not + # one of the rows above it and the table says so. + assert "default: flux-alexis-en (not listed above)" in " ".join( + captured.out.split() + ) + + @patch("deepctl_core.output.get_output_format", return_value="json") + @patch("deepctl_cmd_speak.command.get_output_format", return_value="json") + def test_handle_list_voices_json_carries_canonical_name( + self, + _cmd_fmt, + _core_fmt, + command, + mock_config, + mock_auth_manager, + mock_client, + capsys, + ): + """The serialized JSON carries the same canonical id and languages.""" + mock_client.list_models.return_value = { + "tts": [ + { + "name": "agathe", + "canonical_name": "aura-2-agathe-fr", + "languages": ["fr", "fr-FR"], + } + ] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + command.output_result(result, mock_config) + + payload = json.loads(capsys.readouterr().out) + assert payload["voices"] == [ + { + "name": "aura-2-agathe-fr", + "voice_type": "aura", + "language": "fr, fr-FR", + "languages": ["fr", "fr-FR"], + } + ] + + @patch("deepctl_cmd_speak.command.get_output_format", return_value="default") + def test_handle_list_voices_legacy_catalog_fields( + self, _fmt, command, mock_config, mock_auth_manager, mock_client + ): + """A catalog without the new keys still falls back to name/language.""" + mock_client.list_models.return_value = { + "tts": [{"name": "aura-2-asteria-en", "language": "en"}] + } + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.voices[0].name == "aura-2-asteria-en" + assert result.voices[0].language == "en" + assert result.voices[0].languages == ["en"] + + def test_handle_list_voices_empty( + self, command, mock_config, mock_auth_manager, mock_client + ): + """--list-voices with no TTS voices returns an info result.""" + mock_client.list_models.return_value = {"tts": []} + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + output=None, + model=None, + encoding=None, + container=None, + sample_rate=None, + file=None, + play=False, + list_voices=True, + ) + + assert isinstance(result, SpeakVoicesResult) + assert result.status == "info" + assert result.count == 0 + + def test_handle_list_voices_api_error( + self, command, mock_config, mock_auth_manager, mock_client + ): + """A failing list_models surfaces as an error result, not a traceback.""" + mock_client.list_models.side_effect = Exception("boom") + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + list_voices=True, + ) + + assert result is not None + assert result.status == "error" + assert "boom" in result.message + + # ── --play ─────────────────────────────────────────────────────── + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_streams_wav_to_player( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """flux --play streams a playable WAV into the player as audio arrives.""" + mock_sys.stdin.isatty.return_value = True + # stdout is a TTY and there is no -o: --play is the destination, so the + # "no output specified" guard must not fire. + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + # The empty keep-alive chunk must not reach the player. + mock_client.speak_text_stream.return_value = iter([pcm[0], b"", pcm[1]]) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello from Flux", + output=None, + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + assert result.played is True + assert result.output_path == "" + assert result.bytes_written == 8 + + # What the player actually received: the streaming WAV header this + # command already uses for `| ffplay -`, then the PCM frames. + piped = stub_stdin_player.read_bytes() + assert piped == _streaming_wav_header(sample_rate=24000) + b"".join(pcm) + assert _wav_header_fields(piped) == (24000, 1, 16) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_and_save_together( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + tmp_path, + ): + """--play with -o plays the stream and saves an exact-length WAV.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + mock_client.speak_text_stream.return_value = iter(pcm) + output_file = tmp_path / "out.wav" + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello from Flux", + output=str(output_file), + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.played is True + assert result.output_path == str(output_file) + + # The player got the streaming framing... + assert stub_stdin_player.read_bytes() == ( + _streaming_wav_header(sample_rate=24000) + b"".join(pcm) + ) + # ...and the saved file got a real header a WAV reader can trust. + with wave.open(str(output_file), "rb") as wav: + assert wav.getframerate() == 24000 + assert wav.getnchannels() == 1 + assert wav.getsampwidth() == 2 + assert wav.readframes(wav.getnframes()) == b"".join(pcm) + + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_mulaw_to_file_is_not_wrapped( + self, + mock_sys, + command, + mock_config, + mock_auth_manager, + mock_client, + tmp_path, + ): + """Only linear16 gets a WAV wrapper; mulaw is saved as raw audio.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_client.speak_text_stream.return_value = iter([b"\xff\xfe", b"\xfd"]) + output_file = tmp_path / "out.raw" + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=str(output_file), + model="flux-alexis-en", + encoding="mulaw", + play=False, + ) + + assert isinstance(result, SpeakResult) + assert output_file.read_bytes() == b"\xff\xfe\xfd" + assert result.bytes_written == 3 + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_no_audio_fails( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """An empty Flux stream fails loudly instead of reporting playback.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([]) + + with pytest.raises(click.ClickException, match="returned no audio"): + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_stream_error_is_reported( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """A mid-stream Flux failure exits non-zero rather than half-playing.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + def exploding(): + yield b"\x01\x00" + raise RuntimeError("socket died") + + mock_client.speak_text_stream.return_value = exploding() + + with pytest.raises(click.ClickException, match="Flux streaming failed"): + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_player_exit_code_fails( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + failing_stdin_player, + ): + """A player that exits non-zero fails the command.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([b"\x01\x00\x02\x00"]) + + with pytest.raises(click.ClickException, match="exited with status 3"): + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_player_quit_is_not_an_error( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """Quitting the player mid-playback (broken pipe) is not a failure.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([b"\x01\x00", b"\x02\x00"]) + + @contextmanager + def quit_immediately(player): + sink = Mock() + sink.write.side_effect = BrokenPipeError + yield sink + + with patch("deepctl_cmd_speak.command._player_stdin", quit_immediately): + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + assert result.played is True + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_quit_still_saves_the_whole_file( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + tmp_path, + ): + """Stopping playback must not truncate or skip the -o file.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + mock_client.speak_text_stream.return_value = iter(pcm) + output_file = tmp_path / "out.wav" + + @contextmanager + def quit_immediately(player): + sink = Mock() + sink.write.side_effect = BrokenPipeError + yield sink + + with patch("deepctl_cmd_speak.command._player_stdin", quit_immediately): + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=str(output_file), + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + # The whole utterance still reached the file the user asked for. + with wave.open(str(output_file), "rb") as wav: + assert wav.readframes(wav.getnframes()) == b"".join(pcm) + assert result.bytes_written == output_file.stat().st_size + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_with_redirected_stdout_warns( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + capsys, + ): + """Playing with stdout redirected says why the redirect gets nothing.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = False + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text_stream.return_value = iter([b"\x01\x00"]) + + command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + play=True, + ) + + # Normalize rich's wrapping so the assertion is width-independent. + note = " ".join(capsys.readouterr().err.split()) + assert "--play sends the audio to your player" in note + assert "stdout stays empty" in note + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_flux_play_afplay_gets_a_complete_wav_file( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_file_player, + ): + """afplay has no stdin mode, so it is handed a finished WAV file.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("afplay") + pcm = [b"\x01\x00\x02\x00", b"\x03\x00\x04\x00"] + mock_client.speak_text_stream.return_value = iter(pcm) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello from Flux", + output=None, + model="flux-alexis-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.played is True + # The file afplay was pointed at is a complete, exact-length WAV. + handed_to_player = stub_file_player.copy_path + with wave.open(str(handed_to_player), "rb") as wav: + assert wav.getframerate() == 24000 + assert wav.readframes(wav.getnframes()) == b"".join(pcm) + assert str(stub_file_player.arg_path).endswith(".wav") + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_aura_play_pipes_container_bytes( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """Aura --play pipes the container bytes (mp3) straight to the player.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text.return_value = iter([b"ID3", b"mp3-data"]) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.status == "success" + assert result.played is True + assert result.bytes_written == 11 + assert stub_stdin_player.read_bytes() == b"ID3mp3-data" + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_aura_play_and_save_together( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + tmp_path, + ): + """--play together with -o both saves the file and plays the audio.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text.return_value = iter([b"chunk1", b"chunk2"]) + output_file = tmp_path / "out.mp3" + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=str(output_file), + model="aura-2-asteria-en", + play=True, + ) + + assert isinstance(result, SpeakResult) + assert result.played is True + assert result.output_path == str(output_file) + assert output_file.read_bytes() == b"chunk1chunk2" + assert stub_stdin_player.read_bytes() == b"chunk1chunk2" + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_aura_play_no_audio_errors( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + stub_stdin_player, + ): + """An empty Aura response is an error, not a silent success.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + mock_client.speak_text.return_value = iter([]) + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert result is not None + assert result.status == "error" + assert "no audio" in result.message + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_no_player_found( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + capsys, + ): + """--play with no available player fails clearly before any API call.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.return_value = None + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello world", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert isinstance(result, BaseResult) + assert result.status == "error" + assert "No audio player found" in result.message + mock_client.speak_text.assert_not_called() + mock_client.speak_text_stream.assert_not_called() + # Default output mode prints nothing for a returned result, so the + # command has to put the reason on stderr itself. + printed = " ".join(capsys.readouterr().err.split()) + assert "No audio player found" in printed + + @patch("deepctl_cmd_speak.command.sys") + def test_handle_error_message_keeps_bracketed_paths( + self, + mock_sys, + command, + mock_config, + mock_auth_manager, + mock_client, + capsys, + ): + """A path with brackets prints as typed, not half-eaten as markup.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text=None, + file="/tmp/[draft]/script.txt", + ) + + assert result is not None + assert result.status == "error" + printed = " ".join(capsys.readouterr().err.split()) + assert "/tmp/[draft]/script.txt" in printed + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_rejects_raw_flux_encoding( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """Raw mulaw has no container to detect, so --play refuses it up front.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("ffplay") + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="flux-alexis-en", + encoding="mulaw", + play=True, + ) + + assert result is not None + assert result.status == "error" + assert "raw mulaw" in result.message + mock_client.speak_text_stream.assert_not_called() + + @patch("deepctl_cmd_speak.command.shutil") + @patch("deepctl_cmd_speak.command.sys") + def test_handle_play_rejects_mp3_on_pcm_only_player( + self, + mock_sys, + mock_shutil, + command, + mock_config, + mock_auth_manager, + mock_client, + ): + """aplay cannot decode Aura's mp3, so say so instead of playing noise.""" + mock_sys.stdin.isatty.return_value = True + mock_sys.stdout.isatty.return_value = True + mock_shutil.which.side_effect = _only("aplay") + + result = command.handle( + config=mock_config, + auth_manager=mock_auth_manager, + client=mock_client, + text="Hello", + output=None, + model="aura-2-asteria-en", + play=True, + ) + + assert result is not None + assert result.status == "error" + assert "cannot decode mp3" in result.message + mock_client.speak_text.assert_not_called() + + +class TestAudioPlayerHelpers: + """The player-detection, format, and playback helpers behind --play.""" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_prefers_ffplay(self, mock_shutil): + mock_shutil.which.side_effect = lambda p: f"/usr/bin/{p}" + assert _find_audio_player() == "ffplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_falls_back_in_order(self, mock_shutil): + # ffplay missing, afplay present → afplay wins over paplay/aplay. + mock_shutil.which.side_effect = lambda p: ( + f"/usr/bin/{p}" if p in ("afplay", "paplay", "aplay") else None + ) + assert _find_audio_player() == "afplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_prefers_paplay_over_aplay(self, mock_shutil): + mock_shutil.which.side_effect = lambda p: ( + f"/usr/bin/{p}" if p in ("paplay", "aplay") else None + ) + assert _find_audio_player() == "paplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_last_resort_aplay(self, mock_shutil): + mock_shutil.which.side_effect = _only("aplay") + assert _find_audio_player() == "aplay" + + @patch("deepctl_cmd_speak.command.shutil") + def test_find_audio_player_none(self, mock_shutil): + mock_shutil.which.return_value = None + assert _find_audio_player() is None + + def test_voice_type_badge(self): + assert _voice_type_badge("aura-2-asteria-en") == "aura" + assert _voice_type_badge("Aura-Asteria-EN") == "aura" + assert _voice_type_badge("flux-alexis-en") == "flux" + assert _voice_type_badge("something-else") == "tts" + + def test_play_suffix(self): + assert _play_suffix(is_flux=True, encoding="linear16", container=None) == ".wav" + assert _play_suffix(is_flux=True, encoding=None, container=None) == ".wav" + assert _play_suffix(is_flux=True, encoding="mulaw", container=None) == ".raw" + assert _play_suffix(is_flux=False, encoding="mp3", container=None) == ".mp3" + assert _play_suffix(is_flux=False, encoding=None, container=None) == ".mp3" + assert _play_suffix(is_flux=False, encoding=None, container="wav") == ".wav" + assert _play_suffix(is_flux=False, encoding="opus", container="ogg") == ".ogg" + assert _play_suffix(is_flux=False, encoding="flac", container=None) == ".flac" + # Speak v1 wraps PCM encodings in WAV unless --container none. + assert _play_suffix(is_flux=False, encoding="mulaw", container=None) == ".wav" + assert _play_suffix(is_flux=False, encoding="mulaw", container="none") == ".raw" + + def test_check_playable_accepts_what_players_can_decode(self): + assert ( + _check_playable("ffplay", is_flux=True, encoding=None, container=None) + is None + ) + assert ( + _check_playable("aplay", is_flux=True, encoding="linear16", container=None) + is None + ) + assert ( + _check_playable("ffplay", is_flux=False, encoding=None, container=None) + is None + ) + assert ( + _check_playable("afplay", is_flux=False, encoding="mp3", container=None) + is None + ) + + def test_check_playable_rejects_raw_audio(self): + flux_raw = _check_playable( + "ffplay", is_flux=True, encoding="alaw", container=None + ) + assert flux_raw is not None + assert "raw alaw" in flux_raw + + aura_raw = _check_playable( + "ffplay", is_flux=False, encoding="linear16", container="none" + ) + assert aura_raw is not None + assert "no container" in aura_raw + + def test_check_playable_aplay_rejects_every_compressed_encoding(self): + for encoding in ("mp3", "aac", "opus", "flac"): + problem = _check_playable( + "aplay", is_flux=False, encoding=encoding, container=None + ) + assert problem is not None + assert f"cannot decode {encoding}" in problem + assert "linear16 (WAV)" in problem + + def test_check_playable_paplay_decodes_flac_and_ogg_but_not_mp3_or_aac(self): + """paplay reads FLAC and Ogg through libsndfile; only mp3/aac are out. + + Regression for the review finding that paplay was gated like aplay and + refused FLAC and Opus it can play. + """ + for encoding in ("flac", "opus"): + assert ( + _check_playable( + "paplay", is_flux=False, encoding=encoding, container=None + ) + is None + ) + for encoding in ("mp3", "aac"): + problem = _check_playable( + "paplay", is_flux=False, encoding=encoding, container=None + ) + assert problem is not None + assert f"'paplay' cannot decode {encoding}" in problem + assert "flac" in problem and "opus" in problem + + def test_check_playable_ffplay_and_afplay_accept_every_compressed_encoding(self): + for player in ("ffplay", "afplay"): + for encoding in ("mp3", "aac", "opus", "flac"): + assert ( + _check_playable( + player, is_flux=False, encoding=encoding, container=None + ) + is None + ) + + def test_play_audio_pipes_to_stdin_player(self, stub_stdin_player): + """The bytes handed to a stdin player are exactly the audio.""" + _play_audio("ffplay", b"audio-bytes", suffix=".wav") + + assert stub_stdin_player.read_bytes() == b"audio-bytes" + + def test_play_audio_default_ffplay_argv_reads_stdin(self): + """The real ffplay invocation is the documented no-extra-flags one.""" + assert _STDIN_PLAYER_ARGV["ffplay"] == [ + "ffplay", + "-loglevel", + "error", + "-nodisp", + "-autoexit", + "-", + ] + assert _STDIN_PLAYER_ARGV["aplay"][-1] == "-" + + def test_play_audio_raises_on_player_failure(self, failing_stdin_player): + with pytest.raises(click.ClickException, match="exited with status 3"): + _play_audio("ffplay", b"audio", suffix=".wav") + + def test_play_audio_file_player_gets_the_bytes_and_cleans_up( + self, stub_file_player + ): + """afplay is handed a temp file with the audio, removed afterwards.""" + _play_audio("afplay", b"wav-bytes", suffix=".wav") + + assert stub_file_player.copy_path.read_bytes() == b"wav-bytes" + assert str(stub_file_player.arg_path).endswith(".wav") + # The temp file itself is gone once playback finishes. + assert not Path(stub_file_player.arg_path).exists() + + def test_play_audio_file_player_raises_on_failure(self, failing_file_player): + with pytest.raises(click.ClickException, match="exited with status 4"): + _play_audio("afplay", b"wav-bytes", suffix=".wav") + + def test_player_stdin_kills_the_player_when_the_body_raises(self, tmp_path): + """A failure mid-stream must not leave the player running.""" + script = tmp_path / "sleeper.py" + script.write_text("import sys\nsys.stdin.buffer.read()\n") + with ( + patch.dict(_STDIN_PLAYER_ARGV, {"ffplay": [sys.executable, str(script)]}), + pytest.raises(RuntimeError), + _player_stdin("ffplay"), + ): + raise RuntimeError("stream died") + class TestSpeakResult: """Test cases for SpeakResult model.""" diff --git a/packages/deepctl-core/src/deepctl_core/skill_generator.py b/packages/deepctl-core/src/deepctl_core/skill_generator.py index f6a9e10..d6ade6e 100644 --- a/packages/deepctl-core/src/deepctl_core/skill_generator.py +++ b/packages/deepctl-core/src/deepctl_core/skill_generator.py @@ -591,7 +591,10 @@ def render_developer_guide( lines.append("dg login # Authenticate") lines.append("dg listen audio.wav # Transcribe a file") lines.append("dg listen --mic # Live transcription from mic") - lines.append('dg speak "Hello world" # Text-to-speech') + lines.append( + 'dg speak "Hello from Deepgram" --play # Text-to-speech, played aloud' + ) + lines.append("dg speak --list-voices # List available TTS voices") lines.append("dg projects list # List projects") lines.append("dg usage # View API usage") lines.append("dg mcp # Start MCP server") diff --git a/packages/deepctl-core/tests/unit/test_skill_generator.py b/packages/deepctl-core/tests/unit/test_skill_generator.py index 9074564..8b1e574 100644 --- a/packages/deepctl-core/tests/unit/test_skill_generator.py +++ b/packages/deepctl-core/tests/unit/test_skill_generator.py @@ -172,6 +172,9 @@ def test_contains_cli_section(self): assert "deepctl CLI" in content assert "dg listen" in content assert "dg login" in content + # The speak quickstart leads with playback and voice discovery. + assert 'dg speak "Hello from Deepgram" --play' in content + assert "dg speak --list-voices" in content def test_frontmatter(self): content = render_developer_guide("1.0.0", include_frontmatter=True) diff --git a/web/public/llms-full.txt b/web/public/llms-full.txt index 998a441..5b76e4d 100644 --- a/web/public/llms-full.txt +++ b/web/public/llms-full.txt @@ -144,10 +144,12 @@ ffmpeg -i audio.mp3 -f s16le -ar 16000 -ac 1 - | dg listen --encoding linear16 - ## Text-to-Speech ```bash +dg speak "Hello from Deepgram" --play # Synthesize and play it +dg speak --list-voices # Discover voices dg speak "Hello from Deepgram" -dg speak "Hello" --output output.mp3 -dg speak "Hello" | ffplay -nodisp -autoexit - # Pipe to player -echo "Hello world" | dg speak # Stdin input +dg speak "Hello" --output output.wav # Flux emits WAV; use -m aura-* for mp3 +dg speak "Hello" | ffplay -nodisp -autoexit - # Pipe to player +echo "Hello world" | dg speak # Stdin input dg speak --file script.txt # From file ``` @@ -161,6 +163,8 @@ dg speak --file script.txt # From file | `--encoding` | Audio encoding — Aura: mp3, linear16, flac, mulaw, alaw, opus, aac; Flux: linear16 (default), mulaw, alaw | | `--container` | Audio container — none, wav, ogg (Aura only; Flux auto-wraps linear16 in WAV) | | `--file` | Read text from file | +| `--play` | Play the audio through a local player (ffplay, afplay, paplay, aplay); combine with `--output` to save and play | +| `--list-voices` | List available TTS voices and exit, without generating speech | --- diff --git a/web/public/llms.txt b/web/public/llms.txt index c6a0a80..e144040 100644 --- a/web/public/llms.txt +++ b/web/public/llms.txt @@ -17,7 +17,7 @@ - `dg login` — Authenticate via browser (OIDC device flow) or `--api-key` flag - `dg listen ` — Transcribe files, URLs, microphone (`--mic`), or stdin. Supports `--diarize`, `--webvtt`, `--srt`, `--summarize`. (`dg transcribe` is a hidden alias) -- `dg speak ""` — Text-to-speech synthesis; streams Flux TTS by default (flux-alexis-en), or pick an Aura voice with `-m aura-2-*` +- `dg speak ""` — Text-to-speech synthesis; streams Flux TTS by default (flux-alexis-en), or pick an Aura voice with `-m aura-2-*`. `--play` hears it straight away; `--list-voices` lists the voices - `dg read ` — Text intelligence: sentiment, summaries, topics, intents - `dg models` — List available Deepgram models - `dg projects` — Manage Deepgram projects