From 086a74488852a0aaff5b3d7e2aa5310bc42e1d7a Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 11:04:54 -0400 Subject: [PATCH 01/49] feat(skills): fetch the upstream skill list from the marketplace manifest MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `dg skills install` hardcoded four skill names — api, docs, setup-mcp and starters — so the ten skills deepgram/skills has added since were invisible and the list could only ever go stale. Add skill_bundle, which downloads deepgram/skills as a single tarball and reads .claude-plugin/marketplace.json for the list. That repo's CI validates the manifest against the directories on disk in both directions on every PR, so it is a safe source of truth. One tarball rather than a loop of per-file requests: it gets the manifest, every SKILL.md and every references/ file at one consistent revision, and it cannot half-succeed. Every failure — no network, a bad ref, a malformed manifest, a manifest entry with no directory — raises SkillFetchError rather than returning a subset, because a partial install is indistinguishable from a complete one once it is on disk. Pinned to the released tag deepgram-skills-v1.6.0 for reproducibility, overridable with DEEPCTL_SKILLS_REF. --- .../src/deepctl_core/skill_bundle.py | 349 ++++++++++++++++++ 1 file changed, 349 insertions(+) create mode 100644 packages/deepctl-core/src/deepctl_core/skill_bundle.py diff --git a/packages/deepctl-core/src/deepctl_core/skill_bundle.py b/packages/deepctl-core/src/deepctl_core/skill_bundle.py new file mode 100644 index 0000000..8939ed3 --- /dev/null +++ b/packages/deepctl-core/src/deepctl_core/skill_bundle.py @@ -0,0 +1,349 @@ +"""Fetch the deepgram/skills bundle and expose it as installable skill folders. + +A Deepgram agent skill is a *folder* — ``SKILL.md`` plus whatever supporting +files it ships, notably a ``references/`` subdirectory. The authoritative list +of skills lives in the upstream repo's ``.claude-plugin/marketplace.json``, +which that repo's CI validates against the directories on disk in both +directions on every pull request. Reading that manifest is therefore the only +way to stay in step with upstream without hardcoding a list that goes stale. + +The whole repository is fetched as a single tarball rather than file-by-file: +one request gets the manifest, every ``SKILL.md`` and every ``references/`` +file at a consistent revision, and it cannot half-succeed the way a loop of +per-file requests can. +""" + +from __future__ import annotations + +import json +import os +import shutil +import tarfile +import tempfile +import urllib.error +import urllib.request +from dataclasses import dataclass +from pathlib import Path +from typing import IO, TYPE_CHECKING + +if TYPE_CHECKING: + from collections.abc import Iterator + +__all__ = [ + "DEFAULT_SKILLS_REF", + "RepoSkill", + "SkillFetchError", + "bundle_url", + "fetch_skill_bundle", + "resolve_skills_ref", +] + +# Pinned to a released tag, not a branch: an install of a given deepctl +# version should produce the same skills today and in six months. Override +# with `--ref` or DEEPCTL_SKILLS_REF to track `main` or test a branch. +DEFAULT_SKILLS_REF = "deepgram-skills-v1.6.0" + +SKILLS_REPO = "deepgram/skills" +REF_ENV_VAR = "DEEPCTL_SKILLS_REF" + +_MANIFEST_PATH = ".claude-plugin/marketplace.json" +_PLUGIN_NAME = "deepgram" +_SKILL_ENTRY_FILE = "SKILL.md" +_DOWNLOAD_TIMEOUT = 30 +_MAX_BUNDLE_BYTES = 64 * 1024 * 1024 + + +class SkillFetchError(RuntimeError): + """Raised when the upstream skill bundle cannot be fetched or trusted. + + Always fatal: installing a subset of the skills, or a stale cached + subset, is worse than not installing at all because the user has no way + to tell the difference from a complete install. + """ + + +@dataclass(frozen=True) +class RepoSkill: + """One skill from the upstream repo, as a directory on disk.""" + + name: str + path: Path + + @property + def entry_file(self) -> Path: + """Path to this skill's ``SKILL.md``.""" + return self.path / _SKILL_ENTRY_FILE + + +def resolve_skills_ref(ref: str | None = None) -> str: + """Resolve which upstream ref to install from. + + Precedence: explicit argument, then ``DEEPCTL_SKILLS_REF``, then the + pinned default tag. + """ + if ref: + return ref + from_env = os.environ.get(REF_ENV_VAR, "").strip() + return from_env or DEFAULT_SKILLS_REF + + +def bundle_url(ref: str) -> str: + """Return the codeload tarball URL for ``ref`` (tag, branch or SHA).""" + return f"https://codeload.github.com/{SKILLS_REPO}/tar.gz/{ref}" + + +def fetch_skill_bundle( + ref: str | None = None, + *, + cache_dir: Path | None = None, + force: bool = False, +) -> list[RepoSkill]: + """Download the upstream skill bundle and return its skills. + + Args: + ref: Upstream git ref. Defaults to :func:`resolve_skills_ref`. + cache_dir: Where extracted bundles are kept. Defaults to + ``~/.deepctl/skills/repo_cache``. + force: Re-download even if this ref is already cached. + + Returns: + One :class:`RepoSkill` per entry in the upstream manifest, in + manifest order. + + Raises: + SkillFetchError: The bundle could not be downloaded, unpacked, or + reconciled with its manifest. + """ + resolved = resolve_skills_ref(ref) + root = cache_dir or (Path.home() / ".deepctl" / "skills" / "repo_cache") + target = root / _cache_key(resolved) + + if force or not (target / _MANIFEST_PATH).is_file(): + _download_and_extract(resolved, target) + + return read_manifest_skills(target) + + +def read_manifest_skills(root: Path) -> list[RepoSkill]: + """Read ``.claude-plugin/marketplace.json`` under ``root``. + + Raises: + SkillFetchError: The manifest is missing, malformed, lists no + skills, or names a directory that is not a skill folder. + """ + manifest_path = root / _MANIFEST_PATH + try: + raw = json.loads(manifest_path.read_text(encoding="utf-8")) + except FileNotFoundError: + raise SkillFetchError( + f"Skill manifest {_MANIFEST_PATH} is missing from the {SKILLS_REPO} bundle." + ) + except (OSError, UnicodeDecodeError) as exc: + raise SkillFetchError(f"Could not read {manifest_path}: {exc}") + except json.JSONDecodeError as exc: + raise SkillFetchError( + f"Skill manifest {_MANIFEST_PATH} is not valid JSON: {exc}" + ) + + entries = _manifest_skill_entries(raw) + + skills: list[RepoSkill] = [] + seen: set[str] = set() + for entry in entries: + name = _skill_name(entry) + if name in seen: + raise SkillFetchError(f"Skill manifest lists {name!r} more than once.") + seen.add(name) + path = _resolve_skill_dir(root, entry) + skills.append(RepoSkill(name=name, path=path)) + + return skills + + +# --------------------------------------------------------------------------- +# Manifest parsing +# --------------------------------------------------------------------------- + + +def _manifest_skill_entries(raw: object) -> list[str]: + """Pull ``plugins[deepgram].skills`` out of a parsed manifest.""" + if not isinstance(raw, dict): + raise SkillFetchError("Skill manifest is not a JSON object.") + + plugins = raw.get("plugins") + if not isinstance(plugins, list) or not plugins: + raise SkillFetchError("Skill manifest has no 'plugins' array.") + + entries: object = None + found = False + for candidate in plugins: + if isinstance(candidate, dict) and candidate.get("name") == _PLUGIN_NAME: + entries = candidate.get("skills") + found = True + break + if not found: + raise SkillFetchError(f"Skill manifest has no plugin named {_PLUGIN_NAME!r}.") + + if not isinstance(entries, list) or not entries: + raise SkillFetchError( + f"Plugin {_PLUGIN_NAME!r} in the skill manifest lists no skills." + ) + if not all(isinstance(e, str) and e.strip() for e in entries): + raise SkillFetchError( + f"Plugin {_PLUGIN_NAME!r} in the skill manifest has a " + "non-string skill entry." + ) + return [str(e) for e in entries] + + +def _skill_name(entry: str) -> str: + """Derive a skill's directory name from a manifest entry.""" + name = entry.strip().strip("/").rsplit("/", 1)[-1] + if not name or name in {".", ".."}: + raise SkillFetchError(f"Skill manifest entry {entry!r} has no name.") + return name + + +def _resolve_skill_dir(root: Path, entry: str) -> Path: + """Resolve a manifest entry to a skill directory inside ``root``.""" + relative = _relative_parts(entry) + if relative is None: + raise SkillFetchError( + f"Skill manifest entry {entry!r} escapes the bundle root." + ) + + path = root.joinpath(*relative) + if not path.is_dir(): + raise SkillFetchError( + f"Skill manifest lists {entry!r} but that directory is not in " + f"the {SKILLS_REPO} bundle." + ) + if not (path / _SKILL_ENTRY_FILE).is_file(): + raise SkillFetchError( + f"Skill manifest lists {entry!r} but it has no {_SKILL_ENTRY_FILE}." + ) + return path + + +def _relative_parts(entry: str) -> list[str] | None: + """Split a manifest entry into safe relative path parts, or None.""" + parts: list[str] = [] + for part in entry.strip().split("/"): + if part in ("", "."): + continue + if part == ".." or part.startswith("/"): + return None + parts.append(part) + return parts or None + + +# --------------------------------------------------------------------------- +# Download + extraction +# --------------------------------------------------------------------------- + + +def _cache_key(ref: str) -> str: + """Filesystem-safe directory name for a ref.""" + return "".join(c if c.isalnum() or c in "-._" else "_" for c in ref) + + +def _download_and_extract(ref: str, target: Path) -> None: + """Download the bundle for ``ref`` and replace ``target`` with it.""" + url = bundle_url(ref) + with tempfile.TemporaryDirectory(prefix="deepctl-skills-") as tmp: + tmp_path = Path(tmp) + archive = tmp_path / "bundle.tar.gz" + _download(url, ref, archive) + + unpacked = tmp_path / "unpacked" + unpacked.mkdir() + _extract(archive, unpacked, ref) + + roots = [p for p in unpacked.iterdir() if p.is_dir()] + if len(roots) != 1: + raise SkillFetchError( + f"The {SKILLS_REPO}@{ref} bundle does not have the expected " + "single top-level directory." + ) + + # Validate before publishing to the cache, so a bad bundle never + # replaces a good one. + read_manifest_skills(roots[0]) + + target.parent.mkdir(parents=True, exist_ok=True) + if target.exists(): + shutil.rmtree(target) + shutil.move(str(roots[0]), str(target)) + + +def _download(url: str, ref: str, dest: Path) -> None: + """Fetch ``url`` into ``dest``, mapping every failure to SkillFetchError.""" + try: + with urllib.request.urlopen(url, timeout=_DOWNLOAD_TIMEOUT) as resp: + _copy_limited(resp, dest) + except urllib.error.HTTPError as exc: + if exc.code == 404: + raise SkillFetchError( + f"{SKILLS_REPO} has no ref {ref!r} (HTTP 404 from {url}). " + "Check the --ref value." + ) + raise SkillFetchError( + f"Could not download {SKILLS_REPO}@{ref}: HTTP {exc.code} from {url}." + ) + except (urllib.error.URLError, OSError, ValueError) as exc: + raise SkillFetchError( + f"Could not download {SKILLS_REPO}@{ref} from {url}: {exc}" + ) + + +def _copy_limited(src: IO[bytes], dest: Path) -> None: + """Stream ``src`` to ``dest``, refusing an implausibly large bundle.""" + total = 0 + with dest.open("wb") as fh: + while chunk := src.read(64 * 1024): + total += len(chunk) + if total > _MAX_BUNDLE_BYTES: + raise SkillFetchError( + f"The {SKILLS_REPO} bundle exceeded " + f"{_MAX_BUNDLE_BYTES} bytes; refusing to unpack it." + ) + fh.write(chunk) + + +def _extract(archive: Path, dest: Path, ref: str) -> None: + """Unpack ``archive`` into ``dest``, rejecting unsafe members.""" + try: + with tarfile.open(archive, "r:gz") as tar: + tar.extractall(dest, members=_safe_members(tar, dest)) + except SkillFetchError: + raise + except (tarfile.TarError, OSError, EOFError) as exc: + raise SkillFetchError( + f"The {SKILLS_REPO}@{ref} download is not a readable tar.gz archive: {exc}" + ) + + +def _safe_members(tar: tarfile.TarFile, dest: Path) -> Iterator[tarfile.TarInfo]: + """Yield only regular files and directories that stay inside ``dest``. + + ``tarfile``'s ``filter="data"`` argument is not available on every + Python version this CLI supports, so the checks are explicit. + """ + root = dest.resolve() + for member in tar: + if not (member.isfile() or member.isdir()): + # Symlinks, hardlinks and devices have no place in a skill + # bundle and are how tar extraction turns into arbitrary writes. + continue + name = member.name + if name.startswith("/") or ".." in Path(name).parts: + raise SkillFetchError( + f"Refusing to unpack {name!r}: it escapes the bundle root." + ) + resolved = (root / name).resolve() + if resolved != root and root not in resolved.parents: + raise SkillFetchError( + f"Refusing to unpack {name!r}: it escapes the bundle root." + ) + member.mode = 0o755 if member.isdir() else 0o644 + yield member From 72f6953b3041af87632cf66306d4fc4fc7ebbf67 Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 11:06:15 -0400 Subject: [PATCH 02/49] test(skills): cover the bundle fetcher, including every failure path Asserts the full 14-skill list comes from the manifest, that references/ subdirectories survive extraction, and that no-network, HTTP 404, HTTP 5xx, a corrupt archive and five shapes of malformed manifest all raise rather than install a subset. Also checks a failed refresh leaves a good cache in place, and that path-traversal and symlink tar members are refused. --- .../src/deepctl_core/skill_bundle.py | 10 +- .../tests/unit/test_skill_bundle.py | 282 ++++++++++++++++++ 2 files changed, 290 insertions(+), 2 deletions(-) create mode 100644 packages/deepctl-core/tests/unit/test_skill_bundle.py diff --git a/packages/deepctl-core/src/deepctl_core/skill_bundle.py b/packages/deepctl-core/src/deepctl_core/skill_bundle.py index 8939ed3..bb2a5de 100644 --- a/packages/deepctl-core/src/deepctl_core/skill_bundle.py +++ b/packages/deepctl-core/src/deepctl_core/skill_bundle.py @@ -24,7 +24,7 @@ import urllib.request from dataclasses import dataclass from pathlib import Path -from typing import IO, TYPE_CHECKING +from typing import IO, TYPE_CHECKING, Any if TYPE_CHECKING: from collections.abc import Iterator @@ -314,7 +314,13 @@ def _extract(archive: Path, dest: Path, ref: str) -> None: """Unpack ``archive`` into ``dest``, rejecting unsafe members.""" try: with tarfile.open(archive, "r:gz") as tar: - tar.extractall(dest, members=_safe_members(tar, dest)) + # _safe_members already rejects anything that escapes dest; the + # stdlib filter is belt-and-braces where the interpreter has it + # (3.12+, and the backports in 3.10.12 / 3.11.4). + extra: dict[str, Any] = {} + if hasattr(tarfile, "data_filter"): + extra["filter"] = "data" + tar.extractall(dest, members=_safe_members(tar, dest), **extra) except SkillFetchError: raise except (tarfile.TarError, OSError, EOFError) as exc: diff --git a/packages/deepctl-core/tests/unit/test_skill_bundle.py b/packages/deepctl-core/tests/unit/test_skill_bundle.py new file mode 100644 index 0000000..df4b4eb --- /dev/null +++ b/packages/deepctl-core/tests/unit/test_skill_bundle.py @@ -0,0 +1,282 @@ +"""Unit tests for the upstream skill bundle fetcher.""" + +import io +import json +import tarfile +import urllib.error +from pathlib import Path +from unittest.mock import patch + +import pytest +from deepctl_core.skill_bundle import ( + DEFAULT_SKILLS_REF, + REF_ENV_VAR, + SkillFetchError, + bundle_url, + fetch_skill_bundle, + read_manifest_skills, + resolve_skills_ref, +) + +SKILL_NAMES = [ + "speech-to-text", + "text-to-speech", + "voice-agent", + "audio-intelligence", + "text-intelligence", + "browser-agent", + "api", + "docs", + "starters", + "recipes", + "examples", + "cli", + "setup-mcp", + "self-hosted", +] + +# Mirrors the two upstream skills that ship a references/ subdirectory. +SKILLS_WITH_REFERENCES = {"api": ["listen.md", "speak.md"], "self-hosted": ["k8s.md"]} + + +def _manifest(names, plugin_name="deepgram"): + return { + "name": "deepgram-agent-skills", + "plugins": [ + { + "name": plugin_name, + "source": "./", + "skills": [f"./skills/{n}" for n in names], + } + ], + } + + +def _build_repo(root: Path, names=None, manifest=None) -> Path: + """Create a fake deepgram/skills checkout under ``root``.""" + names = SKILL_NAMES if names is None else names + (root / ".claude-plugin").mkdir(parents=True, exist_ok=True) + payload = _manifest(names) if manifest is None else manifest + (root / ".claude-plugin" / "marketplace.json").write_text( + json.dumps(payload) if not isinstance(payload, str) else payload + ) + for name in names: + skill_dir = root / "skills" / name + skill_dir.mkdir(parents=True, exist_ok=True) + (skill_dir / "SKILL.md").write_text( + f"---\nname: {name}\ndescription: Test skill {name}\n---\n\n# {name}\n" + ) + for ref_file in SKILLS_WITH_REFERENCES.get(name, []): + refs = skill_dir / "references" + refs.mkdir(exist_ok=True) + (refs / ref_file).write_text(f"# {name} / {ref_file}\n") + return root + + +def _tarball(source: Path, top="skills-deepgram-skills-v1.6.0") -> bytes: + buf = io.BytesIO() + with tarfile.open(fileobj=buf, mode="w:gz") as tar: + tar.add(source, arcname=top) + return buf.getvalue() + + +class _FakeResponse(io.BytesIO): + """Minimal stand-in for urlopen's context-managed response.""" + + def __enter__(self): + return self + + def __exit__(self, *exc): + self.close() + return False + + +class TestResolveSkillsRef: + def test_defaults_to_the_pinned_tag(self, monkeypatch): + monkeypatch.delenv(REF_ENV_VAR, raising=False) + assert resolve_skills_ref() == DEFAULT_SKILLS_REF + assert DEFAULT_SKILLS_REF.startswith("deepgram-skills-v") + + def test_env_var_overrides_the_default(self, monkeypatch): + monkeypatch.setenv(REF_ENV_VAR, "main") + assert resolve_skills_ref() == "main" + + def test_explicit_ref_wins_over_env(self, monkeypatch): + monkeypatch.setenv(REF_ENV_VAR, "main") + assert resolve_skills_ref("my-branch") == "my-branch" + + def test_blank_env_var_falls_back(self, monkeypatch): + monkeypatch.setenv(REF_ENV_VAR, " ") + assert resolve_skills_ref() == DEFAULT_SKILLS_REF + + def test_bundle_url_uses_the_ref(self): + assert bundle_url("v1.2.3").endswith("/deepgram/skills/tar.gz/v1.2.3") + + +class TestReadManifestSkills: + def test_returns_every_manifest_entry_in_order(self, tmp_path): + skills = read_manifest_skills(_build_repo(tmp_path)) + assert [s.name for s in skills] == SKILL_NAMES + assert len(skills) == 14 + + def test_skill_paths_are_directories_with_an_entry_file(self, tmp_path): + for skill in read_manifest_skills(_build_repo(tmp_path)): + assert skill.path.is_dir() + assert skill.entry_file.is_file() + + def test_missing_manifest(self, tmp_path): + with pytest.raises(SkillFetchError, match="marketplace.json is missing"): + read_manifest_skills(tmp_path) + + def test_malformed_json(self, tmp_path): + _build_repo(tmp_path, manifest="{not json") + with pytest.raises(SkillFetchError, match="not valid JSON"): + read_manifest_skills(tmp_path) + + def test_manifest_without_plugins(self, tmp_path): + _build_repo(tmp_path, manifest={"name": "x"}) + with pytest.raises(SkillFetchError, match="no 'plugins' array"): + read_manifest_skills(tmp_path) + + def test_manifest_without_the_deepgram_plugin(self, tmp_path): + _build_repo(tmp_path, manifest=_manifest(SKILL_NAMES, plugin_name="other")) + with pytest.raises(SkillFetchError, match="no plugin named 'deepgram'"): + read_manifest_skills(tmp_path) + + def test_manifest_with_empty_skill_list(self, tmp_path): + _build_repo(tmp_path, manifest=_manifest([])) + with pytest.raises(SkillFetchError, match="lists no skills"): + read_manifest_skills(tmp_path) + + def test_manifest_with_non_string_entry(self, tmp_path): + payload = _manifest(["api"]) + payload["plugins"][0]["skills"] = [{"path": "./skills/api"}] + _build_repo(tmp_path, names=["api"], manifest=payload) + with pytest.raises(SkillFetchError, match="non-string skill entry"): + read_manifest_skills(tmp_path) + + def test_manifest_entry_without_a_directory(self, tmp_path): + """A manifest/disk mismatch must fail, never silently install a subset.""" + _build_repo(tmp_path, names=["api"], manifest=_manifest(["api", "ghost"])) + with pytest.raises(SkillFetchError, match="'./skills/ghost'"): + read_manifest_skills(tmp_path) + + def test_manifest_entry_without_a_skill_file(self, tmp_path): + _build_repo(tmp_path, names=["api"]) + (tmp_path / "skills" / "api" / "SKILL.md").unlink() + with pytest.raises(SkillFetchError, match="no SKILL.md"): + read_manifest_skills(tmp_path) + + def test_duplicate_manifest_entries(self, tmp_path): + _build_repo(tmp_path, names=["api"], manifest=_manifest(["api", "api"])) + with pytest.raises(SkillFetchError, match="more than once"): + read_manifest_skills(tmp_path) + + def test_traversing_manifest_entry_is_rejected(self, tmp_path): + _build_repo(tmp_path, names=["api"], manifest=_manifest(["../../etc"])) + with pytest.raises(SkillFetchError): + read_manifest_skills(tmp_path) + + +class TestFetchSkillBundle: + def _fetch(self, tmp_path, payload, **kwargs): + cache = tmp_path / "cache" + with patch( + "urllib.request.urlopen", return_value=_FakeResponse(payload) + ) as opener: + skills = fetch_skill_bundle(cache_dir=cache, **kwargs) + return skills, opener, cache + + def test_extracts_all_fourteen_skills(self, tmp_path): + payload = _tarball(_build_repo(tmp_path / "repo")) + skills, _, _ = self._fetch(tmp_path, payload) + assert [s.name for s in skills] == SKILL_NAMES + + def test_preserves_reference_subdirectories(self, tmp_path): + payload = _tarball(_build_repo(tmp_path / "repo")) + skills, _, _ = self._fetch(tmp_path, payload) + by_name = {s.name: s for s in skills} + for name, files in SKILLS_WITH_REFERENCES.items(): + refs = by_name[name].path / "references" + assert refs.is_dir(), f"{name} lost its references/ directory" + assert sorted(p.name for p in refs.iterdir()) == sorted(files) + + def test_requests_the_pinned_tag_by_default(self, tmp_path, monkeypatch): + monkeypatch.delenv(REF_ENV_VAR, raising=False) + payload = _tarball(_build_repo(tmp_path / "repo")) + _, opener, _ = self._fetch(tmp_path, payload) + assert DEFAULT_SKILLS_REF in opener.call_args[0][0] + + def test_second_call_uses_the_cache(self, tmp_path): + payload = _tarball(_build_repo(tmp_path / "repo")) + _, opener, cache = self._fetch(tmp_path, payload) + assert opener.call_count == 1 + with patch("urllib.request.urlopen", side_effect=AssertionError) as second: + skills = fetch_skill_bundle(cache_dir=cache) + assert second.call_count == 0 + assert len(skills) == 14 + + def test_network_failure_raises(self, tmp_path): + with patch( + "urllib.request.urlopen", + side_effect=urllib.error.URLError("Name or service not known"), + ): + with pytest.raises(SkillFetchError, match="Could not download"): + fetch_skill_bundle(cache_dir=tmp_path / "cache") + + def test_unknown_ref_reports_a_404(self, tmp_path): + err = urllib.error.HTTPError( + "https://codeload.github.com/x", 404, "Not Found", {}, None + ) + with patch("urllib.request.urlopen", side_effect=err): + with pytest.raises(SkillFetchError, match="has no ref 'nope'"): + fetch_skill_bundle("nope", cache_dir=tmp_path / "cache") + + def test_server_error_reports_the_status(self, tmp_path): + err = urllib.error.HTTPError( + "https://codeload.github.com/x", 503, "Unavailable", {}, None + ) + with patch("urllib.request.urlopen", side_effect=err): + with pytest.raises(SkillFetchError, match="HTTP 503"): + fetch_skill_bundle(cache_dir=tmp_path / "cache") + + def test_corrupt_archive_raises(self, tmp_path): + with pytest.raises(SkillFetchError, match="not a readable tar.gz"): + self._fetch(tmp_path, b"this is not a tarball") + + def test_malformed_manifest_does_not_replace_a_good_cache(self, tmp_path): + good = _tarball(_build_repo(tmp_path / "repo")) + _, _, cache = self._fetch(tmp_path, good) + + bad_repo = _build_repo(tmp_path / "bad", manifest="{broken") + with patch( + "urllib.request.urlopen", return_value=_FakeResponse(_tarball(bad_repo)) + ): + with pytest.raises(SkillFetchError, match="not valid JSON"): + fetch_skill_bundle(cache_dir=cache, force=True) + + # The previously cached, valid bundle survived the failed refresh. + assert len(fetch_skill_bundle(cache_dir=cache)) == 14 + + def test_absolute_member_is_rejected(self, tmp_path): + buf = io.BytesIO() + with tarfile.open(fileobj=buf, mode="w:gz") as tar: + info = tarfile.TarInfo("/etc/passwd") + info.size = 3 + tar.addfile(info, io.BytesIO(b"bad")) + with pytest.raises(SkillFetchError, match="escapes the bundle root"): + self._fetch(tmp_path, buf.getvalue()) + + def test_symlink_members_are_skipped(self, tmp_path): + """A skill bundle has no business shipping links.""" + repo = _build_repo(tmp_path / "repo") + buf = io.BytesIO() + with tarfile.open(fileobj=buf, mode="w:gz") as tar: + tar.add(repo, arcname="top") + link = tarfile.TarInfo("top/escape") + link.type = tarfile.SYMTYPE + link.linkname = "/etc/passwd" + tar.addfile(link) + skills, _, cache = self._fetch(tmp_path, buf.getvalue()) + assert len(skills) == 14 + assert not (cache / "deepgram-skills-v1.6.0" / "escape").exists() From 1d644c165dc4f03df0e5f11c15de05e9bfa87069 Mon Sep 17 00:00:00 2001 From: Corey Weathers Date: Sun, 20 Sep 2026 11:13:10 -0400 Subject: [PATCH 03/49] fix(skills): install skills as folders, into the directory each tool reads Every destination this command wrote was wrong, and three of them were paths the target tool does not read at all. Destinations, all taken from each tool's own docs rather than from the previous code: Claude Code ~/.claude/commands/deepgram/*.md -> ~/.claude/skills// Codex ~/.codex/instructions.md -> ~/.agents/skills// Gemini CLI ~/.gemini/GEMINI.md -> ~/.gemini/skills// Cursor ~/.cursor/rules/deepctl.mdc -> ~/.cursor/skills// OpenCode ~/.opencode/agents.md -> ~/.config/opencode/skills/ Cline ~/.cline/rules/deepctl.md -> ~/.cline/skills// Amazon Q ~/.amazonq/rules/deepctl.md -> nothing; prints the one-liner Aider ~/.aider.conf.yml read entry -> nothing; prints the one-liner Notes on the three that were not merely the wrong folder: ~/.codex/instructions.md is absent from current Codex docs and source (global instructions are ~/.codex/AGENTS.md); Cursor user-scope rules are a settings-UI feature, so ~/.cursor/rules/deepctl.mdc was read by nothing; and Amazon Q Developer has no skills mechanism at all, so it now gets the "npx skills add deepgram/skills" one-liner instead of a file. Codex targets ~/.agents/skills rather than ~/.codex/skills because Codex's own source marks the latter deprecated. A skill is copied as a whole folder, so skills/api and skills/self-hosted keep their references/ subdirectories instead of losing them to a one-file-per-skill fetch. The misleading "BEGIN deepctl CLI Reference" markers are gone: they never wrapped a CLI reference, they wrapped four concatenated skills. The marker strings survive in one place only, so install and remove can find and clean up what deepctl 0.3.0 and earlier left behind. Install now fails loudly. A network error, a bad ref or a manifest that does not match the directories upstream raises ClickException (exit 1) rather than printing "No skills were installed" and exiting 0. New --ref flag and DEEPCTL_SKILLS_REF pin, both defaulting to the released tag. --- .agents/skills/api/SKILL.md | 331 +++++++++ .agents/skills/api/references/agent.md | 365 +++++++++ .agents/skills/api/references/auth.md | 43 ++ .agents/skills/api/references/listen.md | 338 +++++++++ .agents/skills/api/references/models.md | 78 ++ .agents/skills/api/references/projects.md | 476 ++++++++++++ .agents/skills/api/references/read.md | 57 ++ .agents/skills/api/references/self-hosted.md | 82 +++ .agents/skills/api/references/speak.md | 274 +++++++ .agents/skills/audio-intelligence/SKILL.md | 144 ++++ .agents/skills/browser-agent/SKILL.md | 198 +++++ .agents/skills/cli/SKILL.md | 188 +++++ .agents/skills/docs/SKILL.md | 106 +++ .agents/skills/examples/SKILL.md | 71 ++ .agents/skills/recipes/SKILL.md | 84 +++ .agents/skills/self-hosted/SKILL.md | 183 +++++ .../self-hosted/references/docker-podman.md | 208 ++++++ .../skills/self-hosted/references/hardware.md | 125 ++++ .../self-hosted/references/kubernetes.md | 244 ++++++ .../self-hosted/references/sagemaker.md | 224 ++++++ .agents/skills/setup-mcp/SKILL.md | 317 ++++++++ .agents/skills/speech-to-text/SKILL.md | 136 ++++ .agents/skills/starters/SKILL.md | 195 +++++ .agents/skills/text-intelligence/SKILL.md | 144 ++++ .agents/skills/text-to-speech/SKILL.md | 162 ++++ .agents/skills/voice-agent/SKILL.md | 177 +++++ .claude/skills/api | 1 + .claude/skills/audio-intelligence | 1 + .claude/skills/browser-agent | 1 + .claude/skills/cli | 1 + .claude/skills/docs | 1 + .claude/skills/examples | 1 + .claude/skills/recipes | 1 + .claude/skills/self-hosted | 1 + .claude/skills/setup-mcp | 1 + .claude/skills/speech-to-text | 1 + .claude/skills/starters | 1 + .claude/skills/text-intelligence | 1 + .claude/skills/text-to-speech | 1 + .claude/skills/voice-agent | 1 + .../src/deepctl_cmd_skills/command.py | 304 +++++--- .../src/deepctl_core/skill_generator.py | 697 ++++++++---------- skills-lock.json | 89 +++ 43 files changed, 5584 insertions(+), 470 deletions(-) create mode 100644 .agents/skills/api/SKILL.md create mode 100644 .agents/skills/api/references/agent.md create mode 100644 .agents/skills/api/references/auth.md create mode 100644 .agents/skills/api/references/listen.md create mode 100644 .agents/skills/api/references/models.md create mode 100644 .agents/skills/api/references/projects.md create mode 100644 .agents/skills/api/references/read.md create mode 100644 .agents/skills/api/references/self-hosted.md create mode 100644 .agents/skills/api/references/speak.md create mode 100644 .agents/skills/audio-intelligence/SKILL.md create mode 100644 .agents/skills/browser-agent/SKILL.md create mode 100644 .agents/skills/cli/SKILL.md create mode 100644 .agents/skills/docs/SKILL.md create mode 100644 .agents/skills/examples/SKILL.md create mode 100644 .agents/skills/recipes/SKILL.md create mode 100644 .agents/skills/self-hosted/SKILL.md create mode 100644 .agents/skills/self-hosted/references/docker-podman.md create mode 100644 .agents/skills/self-hosted/references/hardware.md create mode 100644 .agents/skills/self-hosted/references/kubernetes.md create mode 100644 .agents/skills/self-hosted/references/sagemaker.md create mode 100644 .agents/skills/setup-mcp/SKILL.md create mode 100644 .agents/skills/speech-to-text/SKILL.md create mode 100644 .agents/skills/starters/SKILL.md create mode 100644 .agents/skills/text-intelligence/SKILL.md create mode 100644 .agents/skills/text-to-speech/SKILL.md create mode 100644 .agents/skills/voice-agent/SKILL.md create mode 120000 .claude/skills/api create mode 120000 .claude/skills/audio-intelligence create mode 120000 .claude/skills/browser-agent create mode 120000 .claude/skills/cli create mode 120000 .claude/skills/docs create mode 120000 .claude/skills/examples create mode 120000 .claude/skills/recipes create mode 120000 .claude/skills/self-hosted create mode 120000 .claude/skills/setup-mcp create mode 120000 .claude/skills/speech-to-text create mode 120000 .claude/skills/starters create mode 120000 .claude/skills/text-intelligence create mode 120000 .claude/skills/text-to-speech create mode 120000 .claude/skills/voice-agent create mode 100644 skills-lock.json diff --git a/.agents/skills/api/SKILL.md b/.agents/skills/api/SKILL.md new file mode 100644 index 0000000..b62748a --- /dev/null +++ b/.agents/skills/api/SKILL.md @@ -0,0 +1,331 @@ +--- +name: api +description: > + Deepgram API reference for speech-to-text, text-to-speech, voice agents, audio intelligence, + and account management. Use whenever building with Deepgram APIs — REST or WebSocket. Covers + authentication, all endpoints, query parameters, request/response schemas, and WebSocket + message formats. Reference files are organized by domain: listen (STT — Nova and Flux STT), speak + (TTS — Aura and Flux TTS), agent (voice agents), read (text/audio intelligence), models, + projects, auth, and self-hosted. +--- + +# Deepgram API + +Build with Deepgram's speech-to-text, text-to-speech, voice agent, and audio intelligence APIs. + +> **"Flux" names two separate products.** **Flux STT** is conversational speech-to-text on `/v2/listen` (`model=flux-general-en`). **Flux TTS** is turn-based speech synthesis on `/v2/speak` (`model=flux-{voice}-{language}`). They share a name and a design philosophy — turn-aware, built for voice agents — but they are different endpoints with different models, params, and messages. When a request just says "Flux", check whether it is about transcribing audio or producing it. + +## Getting Started + +All API requests require authentication via API key or JWT: + +- **API Key**: `Authorization: Token ` +- **JWT**: `Authorization: Bearer ` + +Base servers: + +- REST & STT/TTS WebSocket: `https://api.deepgram.com` +- Voice Agent WebSocket **and Voice Agent REST**: `https://agent.deepgram.com` + +Voice Agent's REST endpoints live on the `agent.` host too, not on `api.`: +`GET /v1/agent/settings/think/models` returns 404 on `api.deepgram.com` and 200 on +`agent.deepgram.com`. Everything else REST stays on `api.deepgram.com`. + +### Regional endpoints + +To keep processing inside a geography, swap the host. Same API keys, same paths, same SDKs — +only the base URL changes. Requests are never routed out of region: if the region is +unavailable they fail rather than fall back. + +| Region | Host | +|---|---| +| EU | `api.eu.deepgram.com` | +| Australia | `api.au.deepgram.com` | +| India | `api.in.deepgram.com` | + +**The data plane is regional; the Projects management API is not.** On all three regional hosts: + +| Endpoint | Regional | +|---|---| +| `POST /v1/listen`, `wss://…/v1/listen` | Yes | +| `wss://…/v2/listen` | Yes | +| `POST /v1/speak`, `wss://…/v1/speak` | Yes | +| `POST /v2/speak`, `wss://…/v2/speak` | Yes | +| `POST /v1/read` | Yes | +| `wss://…/v1/agent/converse` | Yes | +| `GET /v1/models` | Yes | +| `POST /v1/auth/grant`, `GET /v1/auth/token` | Yes | +| `/v1/projects/*` (keys, members, usage, billing) | **No — 404** | + +Two host rules that catch people out: + +1. **Voice Agent moves onto the `api.` host regionally.** There is no `agent.eu.deepgram.com` + (the name does not resolve). Use `wss://api.eu.deepgram.com/v1/agent/converse`. The Agent + REST endpoints move with it. Globally it stays on `agent.deepgram.com`. +2. **Keep management calls on `api.deepgram.com`.** Point a client's management calls at a + regional host and `/v1/projects` returns 404, so split the base URL by call type if your + app both transcribes and manages keys. + +Whisper models are not served in any of the three regions — use Nova or Flux STT models there. + +For Deepgram Dedicated and self-hosted hosts, see +[Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints); +for the full per-region feature matrix and SDK snippets, see +[Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints). + +## How Deepgram's APIs Fit Together + +``` + ┌──────────────────────────────┐ + │ api.deepgram.com │ + └──────────────────────────────┘ + │ + ┌───────────┬───────────┬─────┴─────┬───────────┬───────────┐ + ▼ ▼ ▼ ▼ ▼ ▼ + /v1/listen /v2/listen /v1/speak /v2/speak /v1/read /v1/projects/* + Nova — STT Flux — STT Aura — TTS Flux — TTS Text AI Management + REST + WSS WSS only REST + WSS REST + WSS REST only REST only + + ┌──────────────────────────────┐ + │ agent.deepgram.com │ + └──────────────────────────────┘ + │ + ▼ + /v1/agent/converse + WebSocket only + audio ──▶ STT ──▶ LLM ──▶ TTS ──▶ audio + (Deepgram orchestrates the full pipeline) +``` + +## Which API Should I Use? + +``` +Audio → text (transcription)? +├─ General-purpose transcription (captions, batch, call logs, live streams with custom turn logic) +│ └─ Nova models via /v1/listen +│ ├─ Pre-recorded file → REST POST https://api.deepgram.com/v1/listen?model=nova-3 +│ └─ Live stream → WSS wss://api.deepgram.com/v1/listen?model=nova-3 +│ +└─ Conversational audio / voice-agent-style turn detection + └─ Flux STT models via /v2/listen + └─ Live stream → WSS wss://api.deepgram.com/v2/listen?model=flux-general-en + +Text → audio (speech synthesis)? +├─ General-purpose TTS (broadest voice catalog, compressed/containerized audio) +│ └─ Aura models via /v1/speak +│ ├─ One-shot → REST POST https://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ └─ Low-latency stream → WSS wss://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ +└─ Voice-agent TTS (turn-based lifecycle, barge-in, cross-turn consistency) + └─ Flux TTS models via /v2/speak — model is REQUIRED, and must be flux-* + ├─ Pre-render a block → REST POST https://api.deepgram.com/v2/speak?model=flux-alexis-en + └─ Live conversation → WSS wss://api.deepgram.com/v2/speak?model=flux-alexis-en + +Full conversational voice agent (audio in, audio out)? +└─ WSS wss://agent.deepgram.com/v1/agent/converse + Deepgram handles STT + your configured LLM + TTS internally + +Analyze text for insights? +└─ REST POST /v1/read + (summaries, sentiment, topics, intents) +``` + +## Speech-to-Text: Nova (`/v1/listen`) vs Flux STT (`/v2/listen`) + +Both model families are actively maintained and industry-leading. They solve different problems — pick the one that matches your use case. + +| | Nova (`/v1/listen`) | Flux STT (`/v2/listen`) | +|---|---|---| +| Endpoint | `/v1/listen` | `/v2/listen` | +| Available models | `nova-3` (also `nova-3-medical`, `nova-3-pharma`), `nova-2`, `nova`, `enhanced`, `base` | `flux-general-en`, `flux-general-multi` | +| Best for | General transcription — captions, subtitles, call logs, batch | Conversational audio — voice agents, interactive assistants, turn-taking UIs | +| Output | Continuous transcript stream | Structured turn events + transcripts (built-in turn state machine) | +| Turn detection | Manual (`utterance_end_ms`, VAD events) | Built-in (EOT, eager-EOT, turn_index) | +| Transports | REST + WebSocket | WebSocket only | +| Intelligence overlays | Yes — `summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`, etc. | No — smaller focused param set; no `smart_format` / `diarize_model` / `punctuate` | +| Mid-session reconfig | No (reconnect to change) | Yes (`Configure` message updates EOT thresholds + keyterms live) | + +**Pick Nova (`/v1/listen`, `model=nova-3`) when:** +- Generating captions, subtitles, or transcripts for recorded media +- Running batch transcription over files (REST) +- You need analytics overlays (`summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`) +- You want WebSocket streaming with your own turn-detection logic + +**Pick Flux STT (`/v2/listen`, `model=flux-general-en`) when:** +- Building an interactive voice agent or assistant +- You want end-of-turn detection handled for you +- You need low-latency turn signals and barge-in support +- You want to update EOT thresholds or keyterms mid-session without reconnecting + +Migrating from Nova 3 to Flux STT? See the official [Nova 3 → Flux migration guide](https://developers.deepgram.com/docs/flux/nova-3-migration). + +## Text-to-Speech: Aura (`/v1/speak`) vs Flux TTS (`/v2/speak`) + +Both TTS families are actively maintained. `/v2/speak` is a **new endpoint, not a replacement** — `/v1/speak` is unchanged, and there is no aliasing, redirect, or deprecation. The families do not overlap: Aura voices are served only on `/v1/speak`, Flux TTS voices only on `/v2/speak`. + +| | Aura (`/v1/speak`) | Flux TTS (`/v2/speak`) | +|---|---|---| +| Endpoint | `/v1/speak` | `/v2/speak` | +| Models | `aura-2-*` (en, es, de, nl, fr, it, ja), `aura-*` | `flux-{voice}-{language}`, e.g. `flux-alexis-en` — English at launch | +| `model` param | Optional (defaults to `aura-asteria-en`) | **Required**; an `aura-*` string is rejected | +| Best for | Broadest voice catalog, multilingual, compressed audio, one-shot synthesis | Voice agents — streaming LLM output, barge-in, multi-turn conversations | +| Mental model | Text buffer → audio stream | Streaming-first, turn-based conversation | +| Turn lifecycle | None | `SpeechStarted` → audio → `Flushed` → `SpeechMetadata` per turn (server-assigned `speech_id`) | +| Cross-turn context | None (reconnect to reset) | Prosody persists across turns automatically — no API surface | +| Transports | REST + WebSocket | REST (batch) + WebSocket (streaming) | +| Streaming encodings | `linear16`, `mulaw`, `alaw` | `linear16`, `mulaw`, `alaw` — raw audio only | +| Batch encodings | `mp3`, `opus`, `flac`, `aac`, `linear16`, `mulaw`, `alaw` + `container` / `bit_rate` | Same — but batch-only; the socket rejects them | +| Interruption | `Clear` discards the buffer, no feedback | `Interrupt` → `SpeechInterrupted` with `text_spoken` / `text_remaining` | +| Mid-stream reconfig | No (fixed at connection) | Yes — `Configure` updates `speed` only | +| `speed` | `0.7`–`1.5` — Aura-2, English and Spanish only | `0.5`–`1.5` in `0.05` steps | +| `expressivity` | Not supported | `-2`…`2`, default `0` (beta; fixed for the connection) | +| Voice Agent `provider.version` | `v1` (the default when a provider is specified) | `v2` (required) | + +**Pick Aura (`/v1/speak`) when:** +- You need a language other than English, or a specific Aura voice +- You want compressed or containerized output (`mp3`, `opus`, `flac`, `aac`) from a stream +- You're doing one-shot synthesis and don't need a turn lifecycle +- You're already on Aura and nothing in Flux TTS is pulling you over — v1 is unchanged + +**Pick Flux TTS (`/v2/speak`) when:** +- Building a voice agent, phone assistant, or customer-service bot +- You're streaming LLM tokens to a speaker in real time and want the lowest time-to-first-audio +- The user may barge in mid-response and you need to know what they actually heard +- You want tone to carry across turns without managing state yourself +- You're pre-rendering fixed audio (IVR prompts, notifications) with a Flux TTS voice — use the batch REST transport + +Migrating from Aura? See the official [Migrating from Aura to Flux TTS](https://developers.deepgram.com/docs/flux-tts/migrating) guide and [Batch vs Streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming). + +## API Domains + +| Domain | REST | WebSocket | Reference | +|--------|------|-----------|-----------| +| Listen v1 — STT, Nova models | `POST /v1/listen` | `wss://api.deepgram.com/v1/listen` | [listen.md](references/listen.md) | +| Listen v2 — STT, Flux STT (conversational) | — | `wss://api.deepgram.com/v2/listen` | [listen.md](references/listen.md) | +| Speak v1 — TTS, Aura models | `POST /v1/speak` | `wss://api.deepgram.com/v1/speak` | [speak.md](references/speak.md) | +| Speak v2 — TTS, Flux TTS (turn-based) | `POST /v2/speak` | `wss://api.deepgram.com/v2/speak` | [speak.md](references/speak.md) | +| Voice Agent | `GET agent.deepgram.com/v1/agent/settings/think/models` | `wss://agent.deepgram.com/v1/agent/converse` | [agent.md](references/agent.md) | +| Read (Intelligence) | `POST /v1/read` | — | [read.md](references/read.md) | +| Models | `GET /v1/models` | — | [models.md](references/models.md) | +| Projects | `/v1/projects/*` | — | [projects.md](references/projects.md) | +| Auth | `POST /v1/auth/grant` | — | [auth.md](references/auth.md) | +| Self-Hosted | `/v1/projects/*/self-hosted/*` | — | [self-hosted.md](references/self-hosted.md) | + +## Common Mistakes to Avoid + +### All APIs + +1. **Feature flags are query params — except for Voice Agent and the v2 mid-session updates.** For `/v1/listen`, `/v2/listen`, `/v1/speak`, and `/v2/speak`, initial options go on the URL. The request body carries only audio data (REST) or audio frames (WebSocket). Exceptions: `/v1/agent/converse` has no URL query params at all (all config goes in the `Settings` message); `/v2/listen` supports a `Configure` message after connection to update EOT thresholds and keyterms mid-session; and `/v2/speak` supports a `Configure` message that updates `speed` only. Also note that `/v2/listen` has a much smaller param set than `/v1/listen` — flags like `smart_format`, `diarize_model`, and `punctuate` are not available. + +2. **Rate limits are concurrent connections, not total requests.** A 429 means too many simultaneous open connections, not too high a request volume. Diarization and other compute-heavy features reduce your concurrency allowance further. + +### STT WebSocket (`/v1/listen`) + +3. **Send KeepAlive as a text frame, not binary.** The connection closes after 10 seconds of no audio. Send `{"type":"KeepAlive"}` as a text (JSON) frame every 3–5 seconds during silence. Sending it as a binary frame causes transcription delays — the audio pipeline chokes — not a silent no-op. + +4. **Never send empty byte payloads.** Sending a zero-length binary frame to `/v1/listen` is treated as a close — it terminates the connection. Always check that your audio packet has length before sending. + +5. **`encoding` must match the actual audio format.** If `encoding=linear16` but you're sending opus, you'll get a DATA-0000 error or garbled output. Omit `encoding` entirely when sending containerized formats (mp3, wav, ogg) — Deepgram detects them automatically. + +6. **Timestamps reset on reconnect.** Each new WebSocket connection restarts timestamps at 00:00:00. For real-time apps, maintain a timestamp offset across reconnections or you'll silently corrupt your transcript timeline. + +### TTS WebSocket (`/v1/speak`) + +7. **Don't send empty text.** A `Speak` message with an empty `text` field returns a 400 error. Always validate input before sending. + +8. **Character rate limiting (DATA-0001) means slow down, not retry.** If you hit this, reduce how fast you're submitting text chunks — don't immediately retry or you'll compound the problem. + +### Flux TTS (`/v2/speak`) + +9. **`model` is required, and must be a `flux-*` voice.** Unlike `/v1/speak` there is no default — a connection or request without `model` is rejected. Aura strings are rejected on `/v2/speak`, and Flux voices are not served by `/v1/speak`; the two families never mix. Model strings are `flux-{voice}-{language}`, e.g. `flux-alexis-en`. There is no version segment — generations roll forward behind a stable name, as with Flux STT. + +10. **`Flush` ends the turn — it is not a v1-style buffer flush.** There is no `Finalize`; it's folded into `Flush`. Audio starts streaming on its own before you flush, so don't wait to send text. Use the turn's `SpeechMetadata` (not `Flushed`) as your end-of-turn signal — it arrives once all of the turn's audio has been sent, and carries the billing and timing counts, so you can drop client-side character or duration tracking. The server assigns the turn's `speech_id`; never send one yourself. + +11. **Streaming is raw audio only, and rejects anything it doesn't recognize.** The WebSocket emits non-containerized audio, so `encoding` is limited to `linear16` (default), `mulaw`, or `alaw`. The compressed and containerized encodings (`mp3`, `opus`, `flac`, `aac`) and the `container`, `bit_rate`, `callback`, `callback_method`, and `priority` params are **batch-only** — sending them to the socket fails the connection, as does any unknown or misspelled param. Use the batch REST transport when you need compressed output. + +12. **Insert whitespace between separate generations — the server won't.** Text normalization runs before synthesis, but successive `Speak` messages are concatenated verbatim. Sending `"Hello world."` then `"How are you?"` is processed as `"Hello world.How are you?"`, which causes sentence-boundary artifacts. Add a single space (or the right separator for non-whitespace languages) when you stitch a reply, a tool-call result, and another reply together. Send plain text: SSML and other markup is stripped, with an `INPUT_MARKUP_STRIPPED` warning. + +### Voice Agent (`/v1/agent/converse`) + +13. **Send the `Settings` message before any audio.** The agent ignores everything until it receives and acknowledges the Settings configuration. Message ordering is strictly required. + +14. **`agent.speak.provider.version` selects the TTS family — and omitting `agent.speak` now gives you Flux TTS.** Set `version` to `v2` for Flux TTS or `v1` for Aura; when you specify a provider but omit `version`, it defaults to `v1`. But if you omit `agent.speak` entirely, the agent defaults to Flux TTS with the `flux-kit-en` voice. Switch families by changing `version` and `model` together — a `flux-*` model under `v1`, or an `aura-*` model under `v2`, is invalid: + ```json + { "agent": { "speak": { "provider": { "type": "deepgram", "version": "v2", "model": "flux-alexis-en" } } } } + ``` + +15. **The Voice Agent REST endpoints live on `agent.deepgram.com`, not `api.deepgram.com`.** `GET /v1/agent/settings/think/models` — the list of LLMs you can name in `agent.think.provider` — returns **404 on `api.deepgram.com`** and 200 on `agent.deepgram.com`. Same key, same path; only the host differs, so a client with one hardcoded base URL silently gets a 404 that looks like a missing feature. The three regional `api.*` hosts serve it as well. + +### Flux STT model (`/v2/listen`) + +16. **Use `/v2/listen` and a `flux-general-*` model.** Two are served: `flux-general-en` (English) and `flux-general-multi` (multilingual, and the only model that accepts `language_hint` / `language_hints`). `/v1/listen` does not support Flux STT, and `model=flux` alone is not a valid value. Do not include `language` or `encoding` params for containerized audio. + +17. **Use `Configure` to update EOT thresholds and keyterms mid-session.** Unlike `/v1/listen`, Flux STT supports live reconfiguration after connection — no need to reconnect to change turn detection sensitivity or boost new keyterms: + ```json + { "type": "Configure", "thresholds": { "eot_threshold": "0.8", "eot_timeout_ms": "3000" }, "keyterms": ["Deepgram"] } + ``` + The server responds with `ConfigureSuccess` (echoing back applied values) or `ConfigureFailure`. Omitted threshold fields keep their current values. + +18. **`ForceEndTurn` outside a turn is a `Warning`, not an error — and the socket stays open.** Sending `{"type":"ForceEndTurn"}` while no turn is in progress returns `{"type":"Warning","code":"FORCE_END_TURN_NO_ACTIVE_TURN","description":"Received ForceEndTurn while no turn was active; the request was ignored."}` and the connection continues. Do not treat it as fatal or reconnect. Neither the `Warning` message nor this code is in the AsyncAPI spec yet, so `references/listen.md` cannot show them. When `ForceEndTurn` *does* land mid-turn, the resulting `TurnInfo` carries `event: "EndOfTurn"` with `trigger: "manual"` — `trigger` is `model` | `manual` | `timeout`, it appears on `EndOfTurn` and nowhere else, and it is an open enum, so tolerate values you do not recognize. + +### Nova diarization (`/v1/listen`) + +19. **Use `diarize_model`, and never send it alongside `diarize`.** `diarize` is deprecated. `diarize_model` both enables diarization and picks the version, so you do not also need `diarize=true` — and sending both fails the request: `400 "diarize_model cannot be used together with diarize or diarize_version."`. Values are `latest`, `v1`, and `v2` for batch (`latest` is currently v2), and `latest` or `v1` for streaming. When diarization is on, `metadata.diarize_info` reports which model actually ran (`{"model_uuid": …, "arch": "v2"}`), which is the only way to tell what `latest` resolved to. + +### Text and Audio Intelligence (`/v1/read`, `/v1/listen`) + +20. **`language` is required on `/v1/read`, and it is validated before anything else.** There is no default, despite what `references/read.md` says: omitting it returns `400 INVALID_QUERY_PARAMETER` — "Failed to deserialize query parameters: missing field `language`" — which masks every other problem in the request. English only — `language=multi` is rejected, and `en-US` is accepted but echoed back as `en`. Two more `/v1/read` shapes worth knowing: the JSON body takes **exactly one** of `text` or `url` (both or neither gives `PAYLOAD_ERROR`, and `url` must point at a plain-text document — audio gives `REMOTE_CONTENT_ERROR`), and it is POST-only (`GET` and a WebSocket upgrade both return 405). `summarize` on `/v1/read` accepts `v2` as well as `true`, contrary to the reference. Result paths differ per endpoint: `/v1/read` returns `results.summary.text`, `/v1/listen` returns `results.summary.short`, so code that handles both has to branch. (`sentiment` maps to `results.sentiments` on both.) + +21. **On the Nova streaming socket, only `detect_entities` works — and the other four fail in three different ways.** `detect_entities=true` is supported and puts `entities` at the **top level** of each `Results` message, beside `channel`, not inside `channel.alternatives[0]`. The other four are prerecorded-only: `summarize` fails the handshake with `400 "Summarization is not available for streaming."`; `topics` and `intents` fail it with `403 UNAUTHORIZED_FEATURES_REQUESTED`, which reads like a key-permissions problem even when the same key's prerecorded `topics`/`intents` calls return 200; and `sentiment` is the trap — the handshake succeeds, no error is ever sent, and sentiment simply never appears in the results. + +### Authentication + +22. **JWT TTL applies only to the initial handshake.** Tokens default to 30 seconds. Once the WebSocket connection is established, the token expiring does not close it — tokens are only needed for the upgrade request. + +## SDK-Specific Skills + +This `api` skill covers the product contracts (endpoints, query params, message shapes) that are identical across SDKs. For **language-idiomatic code** — imports, async patterns, builder APIs, common errors — install the SDK-specific skills. Each Deepgram SDK publishes 7 product skills named `deepgram-{lang}-{product}` (e.g. `deepgram-python-speech-to-text`, `deepgram-js-voice-agent`). The `deepgram-{lang}-` prefix avoids collisions when you install skills from multiple SDKs. + +```bash +# Install all skills from a specific SDK +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET + +# Or install a specific product skill from one SDK (note the deepgram-{lang}- prefix) +npx skills add deepgram/deepgram-python-sdk --skill deepgram-python-speech-to-text +npx skills add deepgram/deepgram-js-sdk --skill deepgram-js-voice-agent +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +| Skill | Purpose | +|---|---| +| `recipes` | Minimal runnable snippets per feature per language | +| `examples` | Full integration examples with third-party platforms (Twilio, LiveKit, etc.) | +| `starters` | Runnable starter apps (framework × feature matrix) | +| `docs` | Navigate Deepgram documentation | +| `audio-intelligence` | The `summarize`, `sentiment`, `topics`, `intents`, and `detect_entities` parameters on `/v1/listen` | +| `text-intelligence` | `POST /v1/read` for text you already have | +| `browser-agent` | The Browser Agent SDK packages for running an agent in a browser | +| `cli` | `deepctl` for shell and CI work | +| `self-hosted` | Running Deepgram on your own GPUs | +| `setup-mcp` | Install the Deepgram MCP server | + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) +- [Speech-to-Text Getting Started](https://developers.deepgram.com/docs/stt/getting-started) +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS Models](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Audio Intelligence](https://developers.deepgram.com/docs/audio-intelligence) +- [Self-Hosted Deployments](https://developers.deepgram.com/docs/self-hosted-introduction) +- [Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints) +- [Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints) diff --git a/.agents/skills/api/references/agent.md b/.agents/skills/api/references/agent.md new file mode 100644 index 0000000..cb045db --- /dev/null +++ b/.agents/skills/api/references/agent.md @@ -0,0 +1,365 @@ +# Deepgram Agent API + +Voice Agent — build conversational voice agents. + +## Documentation + +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### GET `/v1/agent/settings/think/models` + +List Agent Think Models + +Retrieves the available think models that can be used for AI agent processing + +#### Responses + +**200**: List of available think models +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/agents` + +List Agent Configurations + +Returns all agent configurations for the specified project. Configurations are returned in their uninterpolated form—template variable placeholders appear as-is rather than with their substituted values. + +#### Responses + +**200**: A list of agent configurations +**400**: Invalid Request + +### POST `/v1/projects/{project_id}/agents` + +Create an Agent Configuration + +Creates a new reusable agent configuration. The `config` field must be a valid JSON string representing the `agent` block of a Settings message. The returned `agent_id` can be passed in place of the full `agent` object in future Settings messages. + +#### Request Body + +**application/json** + +- `config` string **(required)** — A valid JSON string representing the agent block of a Settings message +- `metadata` object — A map of arbitrary key-value pairs for labeling or organizing the agent configuration +- `api_version` integer (default: `1`) — API version. Defaults to 1 + +#### Responses + +**200**: Agent configuration created successfully +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/agents/{agent_id}` + +Get an Agent Configuration + +Returns the specified agent configuration in its uninterpolated form + +#### Responses + +**200**: An agent configuration +**400**: Invalid Request + +### PUT `/v1/projects/{project_id}/agents/{agent_id}` + +Update Agent Metadata + +Updates the metadata associated with an agent configuration. The config itself is immutable—to change the configuration, delete the existing agent and create a new one. + +#### Request Body + +**application/json** + +- `metadata` object **(required)** — A map of string key-value pairs to associate with this agent configuration + +#### Responses + +**200**: Agent configuration updated +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/agents/{agent_id}` + +Delete an Agent Configuration + +Deletes the specified agent configuration. Deleting an agent configuration can cause a production outage if your service references this agent UUID. Migrate all active sessions to a new configuration before deleting. + +#### Responses + +**200**: Agent configuration deleted +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/agent-variables` + +List Agent Variables + +Returns all template variables for the specified project + +#### Responses + +**200**: A list of agent variables +**400**: Invalid Request + +### POST `/v1/projects/{project_id}/agent-variables` + +Create an Agent Variable + +Creates a new template variable. Variables follow the `DG_` naming format and can substitute any JSON value in an agent configuration. + +#### Request Body + +**application/json** + +- `key` string **(required)** — The variable name, following the DG_ format +- `value` any **(required)** — The value to substitute. Can be any valid JSON type (string, number, boolean, object, or array) +- `api_version` integer (default: `1`) — API version. Defaults to 1 + +#### Responses + +**200**: Agent variable created successfully +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/agent-variables/{variable_id}` + +Get an Agent Variable + +Returns the specified template variable + +#### Responses + +**200**: An agent variable +**400**: Invalid Request + +### PATCH `/v1/projects/{project_id}/agent-variables/{variable_id}` + +Update an Agent Variable + +Updates the value of an existing template variable + +#### Request Body + +**application/json** + +- `value` any **(required)** — The new value to substitute + +#### Responses + +**200**: Agent variable updated +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/agent-variables/{variable_id}` + +Delete an Agent Variable + +Deletes the specified template variable + +#### Responses + +**200**: Agent variable deleted +**400**: Invalid Request + +## WebSocket API + +### WebSocket `/v1/agent/converse` +> Server: `wss://agent.deepgram.com` + +Build a conversational voice agent using Deepgram's Voice Agent WebSocket + +#### Client → Server Messages + +**AgentV1Settings** — Send settings configuration to Deepgram's Voice Agent API + + - `type` `Settings` **(required)** + - `tags` string[] — Tags to associate with the request + - `experimental` boolean (default: `false`) — To enable experimental features + - `flags` { history: boolean } + - `mip_opt_out` boolean (default: `false`) — To opt out of Deepgram Model Improvement Program + - `audio` { input: { encoding: `linear16` | `linear32` | `flac` | `alaw` | `mulaw` | `amr-nb` | `amr-wb` | `opus` | `ogg-opus` | `speex` | `g729`, sample_rate: integer }, output: { encoding: `linear16` | `mulaw` | `alaw` | `mp3` | `opus` | `flac` | `aac`, sample_rate: integer, bitrate: integer, container: `none` | `wav` | `ogg` } } **(required)** + - `agent` { language: string, context: { messages: object | object[] }, listen: { provider: object | object }, think: { provider: object | object | object | object | object, endpoint: object, functions: object[], prompt: string, context_length: `max` | number } | { provider: object | object | object | object | object, endpoint: object, functions: object[], prompt: string, context_length: `max` | number }[], speak: { provider: object | object | object | object | object, endpoint: object } | { provider: object | object | object | object | object, endpoint: object }[], greeting: string } | string **(required)** + +**AgentV1UpdateListen** — Send update listen to Deepgram's Voice Agent API + + - `type` `UpdateListen` **(required)** — Message type identifier for updating the listen configuration + - `listen` { provider: { type: `deepgram`, version: `v1`, model: string, language: string, keyterms: string[], smart_format: boolean } | { type: `deepgram`, version: `v2`, model: string, language_hints: string[], eot_threshold: number, eager_eot_threshold: number, eot_timeout_ms: integer, keyterms: string[] } } **(required)** — Listen configuration to update. Contains a provider object with the same schema as Settings. The model and language can be changed mid-session. Keyterms can only be updated mid-session for Flux models. + +**AgentV1UpdateThink** — Send update think to Deepgram's Voice Agent API + + - `type` `UpdateThink` **(required)** — Message type identifier for updating the think model + - `think` { provider: { type: `open_ai`, version: `v1`, model: `gpt-5` | `gpt-5-mini` | `gpt-5-nano` | `gpt-4.1` | `gpt-4.1-mini` | `gpt-4.1-nano` | `gpt-4o` | `gpt-4o-mini`, temperature: number, reasoning_mode: `none` | `minimal` | `low` | `medium` | `high` } | { type: `aws_bedrock`, model: `anthropic/claude-3-5-sonnet-20240620-v1:0` | `anthropic/claude-3-5-haiku-20240307-v1:0`, temperature: number, credentials: object } | { type: `anthropic`, version: `v1`, model: `claude-3-5-haiku-latest` | `claude-sonnet-4-20250514`, temperature: number } | { type: `google`, version: `ai-studio-v1beta` | `gemini-enterprise-agent-v1` | `v1beta`, model: `gemini-2.0-flash` | `gemini-2.0-flash-lite` | `gemini-2.5-flash`, temperature: number } | { type: `groq`, version: `v1`, model: `openai/gpt-oss-20b`, temperature: number, reasoning_mode: `none` | `minimal` | `low` | `medium` | `high` }, endpoint: { url: string, headers: object }, functions: { name: string, description: string, parameters: object, defer_until_eot: boolean, endpoint: object }[], prompt: string, context_length: `max` | number } | { provider: { type: `open_ai`, version: `v1`, model: `gpt-5` | `gpt-5-mini` | `gpt-5-nano` | `gpt-4.1` | `gpt-4.1-mini` | `gpt-4.1-nano` | `gpt-4o` | `gpt-4o-mini`, temperature: number, reasoning_mode: `none` | `minimal` | `low` | `medium` | `high` } | { type: `aws_bedrock`, model: `anthropic/claude-3-5-sonnet-20240620-v1:0` | `anthropic/claude-3-5-haiku-20240307-v1:0`, temperature: number, credentials: object } | { type: `anthropic`, version: `v1`, model: `claude-3-5-haiku-latest` | `claude-sonnet-4-20250514`, temperature: number } | { type: `google`, version: `ai-studio-v1beta` | `gemini-enterprise-agent-v1` | `v1beta`, model: `gemini-2.0-flash` | `gemini-2.0-flash-lite` | `gemini-2.5-flash`, temperature: number } | { type: `groq`, version: `v1`, model: `openai/gpt-oss-20b`, temperature: number, reasoning_mode: `none` | `minimal` | `low` | `medium` | `high` }, endpoint: { url: string, headers: object }, functions: { name: string, description: string, parameters: object, defer_until_eot: boolean, endpoint: object }[], prompt: string, context_length: `max` | number }[] **(required)** + +**AgentV1UpdateSpeak** — Send update speak to Deepgram's Voice Agent API + + - `type` `UpdateSpeak` **(required)** — Message type identifier for updating the speak model + - `speak` { provider: { type: `deepgram`, version: string, model: `aura-asteria-en` | `aura-luna-en` | `aura-stella-en` | `aura-athena-en` | `aura-hera-en` | `aura-orion-en` | `aura-arcas-en` | `aura-perseus-en` | `aura-angus-en` | `aura-orpheus-en` | `aura-helios-en` | `aura-zeus-en` | `aura-2-amalthea-en` | `aura-2-andromeda-en` | `aura-2-apollo-en` | `aura-2-arcas-en` | `aura-2-aries-en` | `aura-2-asteria-en` | `aura-2-athena-en` | `aura-2-atlas-en` | `aura-2-aurora-en` | `aura-2-callista-en` | `aura-2-cora-en` | `aura-2-cordelia-en` | `aura-2-delia-en` | `aura-2-draco-en` | `aura-2-electra-en` | `aura-2-harmonia-en` | `aura-2-helena-en` | `aura-2-hera-en` | `aura-2-hermes-en` | `aura-2-hyperion-en` | `aura-2-iris-en` | `aura-2-janus-en` | `aura-2-juno-en` | `aura-2-jupiter-en` | `aura-2-luna-en` | `aura-2-mars-en` | `aura-2-minerva-en` | `aura-2-neptune-en` | `aura-2-odysseus-en` | `aura-2-ophelia-en` | `aura-2-orion-en` | `aura-2-orpheus-en` | `aura-2-pandora-en` | `aura-2-phoebe-en` | `aura-2-pluto-en` | `aura-2-saturn-en` | `aura-2-selene-en` | `aura-2-thalia-en` | `aura-2-theia-en` | `aura-2-vesta-en` | `aura-2-zeus-en` | `aura-2-sirio-es` | `aura-2-nestor-es` | `aura-2-carina-es` | `aura-2-celeste-es` | `aura-2-alvaro-es` | `aura-2-diana-es` | `aura-2-aquila-es` | `aura-2-selena-es` | `aura-2-estrella-es` | `aura-2-javier-es` | `flux-alexis-en` | `flux-bree-en` | `flux-brittany-en` | `flux-brooke-en` | `flux-bruce-en` | `flux-cliff-en` | `flux-cole-en` | `flux-colin-en` | `flux-conor-en` | `flux-donovan-en` | `flux-drew-en` | `flux-elise-en` | `flux-gemma-en` | `flux-haley-en` | `flux-hannah-en` | `flux-heather-en` | `flux-jack-en` | `flux-kai-en` | `flux-kelsey-en` | `flux-kit-en` | `flux-maeve-en` | `flux-marcelo-en` | `flux-marcus-en` | `flux-meena-en` | `flux-meghan-en` | `flux-miles-en` | `flux-naveen-en` | `flux-paige-en` | `flux-priya-en` | `flux-rufus-en` | `flux-sean-en` | `flux-sharon-en` | `flux-sienna-en` | `flux-tanner-en` | `flux-wade-en` | `flux-wes-en`, speed: number, expressivity: `-2` | `-1` | `0` | `1` | `2` } | { type: `eleven_labs`, version: `v1`, model_id: `eleven_turbo_v2_5` | `eleven_monolingual_v1` | `eleven_multilingual_v2`, language: string, language_code: string } | { type: `cartesia`, version: `2025-03-17`, model_id: `sonic-2` | `sonic-multilingual`, voice: object, language: string, volume: number } | { type: `open_ai`, version: `v1`, model: `tts-1` | `tts-1-hd`, voice: `alloy` | `echo` | `fable` | `onyx` | `nova` | `shimmer` } | { type: `aws_polly`, voice: `Matthew` | `Joanna` | `Amy` | `Emma` | `Brian` | `Arthur` | `Aria` | `Ayanda`, language: string, language_code: string, engine: `generative` | `long-form` | `standard` | `neural`, credentials: object }, endpoint: { url: string, headers: object } } | { provider: { type: `deepgram`, version: string, model: `aura-asteria-en` | `aura-luna-en` | `aura-stella-en` | `aura-athena-en` | `aura-hera-en` | `aura-orion-en` | `aura-arcas-en` | `aura-perseus-en` | `aura-angus-en` | `aura-orpheus-en` | `aura-helios-en` | `aura-zeus-en` | `aura-2-amalthea-en` | `aura-2-andromeda-en` | `aura-2-apollo-en` | `aura-2-arcas-en` | `aura-2-aries-en` | `aura-2-asteria-en` | `aura-2-athena-en` | `aura-2-atlas-en` | `aura-2-aurora-en` | `aura-2-callista-en` | `aura-2-cora-en` | `aura-2-cordelia-en` | `aura-2-delia-en` | `aura-2-draco-en` | `aura-2-electra-en` | `aura-2-harmonia-en` | `aura-2-helena-en` | `aura-2-hera-en` | `aura-2-hermes-en` | `aura-2-hyperion-en` | `aura-2-iris-en` | `aura-2-janus-en` | `aura-2-juno-en` | `aura-2-jupiter-en` | `aura-2-luna-en` | `aura-2-mars-en` | `aura-2-minerva-en` | `aura-2-neptune-en` | `aura-2-odysseus-en` | `aura-2-ophelia-en` | `aura-2-orion-en` | `aura-2-orpheus-en` | `aura-2-pandora-en` | `aura-2-phoebe-en` | `aura-2-pluto-en` | `aura-2-saturn-en` | `aura-2-selene-en` | `aura-2-thalia-en` | `aura-2-theia-en` | `aura-2-vesta-en` | `aura-2-zeus-en` | `aura-2-sirio-es` | `aura-2-nestor-es` | `aura-2-carina-es` | `aura-2-celeste-es` | `aura-2-alvaro-es` | `aura-2-diana-es` | `aura-2-aquila-es` | `aura-2-selena-es` | `aura-2-estrella-es` | `aura-2-javier-es` | `flux-alexis-en` | `flux-bree-en` | `flux-brittany-en` | `flux-brooke-en` | `flux-bruce-en` | `flux-cliff-en` | `flux-cole-en` | `flux-colin-en` | `flux-conor-en` | `flux-donovan-en` | `flux-drew-en` | `flux-elise-en` | `flux-gemma-en` | `flux-haley-en` | `flux-hannah-en` | `flux-heather-en` | `flux-jack-en` | `flux-kai-en` | `flux-kelsey-en` | `flux-kit-en` | `flux-maeve-en` | `flux-marcelo-en` | `flux-marcus-en` | `flux-meena-en` | `flux-meghan-en` | `flux-miles-en` | `flux-naveen-en` | `flux-paige-en` | `flux-priya-en` | `flux-rufus-en` | `flux-sean-en` | `flux-sharon-en` | `flux-sienna-en` | `flux-tanner-en` | `flux-wade-en` | `flux-wes-en`, speed: number, expressivity: `-2` | `-1` | `0` | `1` | `2` } | { type: `eleven_labs`, version: `v1`, model_id: `eleven_turbo_v2_5` | `eleven_monolingual_v1` | `eleven_multilingual_v2`, language: string, language_code: string } | { type: `cartesia`, version: `2025-03-17`, model_id: `sonic-2` | `sonic-multilingual`, voice: object, language: string, volume: number } | { type: `open_ai`, version: `v1`, model: `tts-1` | `tts-1-hd`, voice: `alloy` | `echo` | `fable` | `onyx` | `nova` | `shimmer` } | { type: `aws_polly`, voice: `Matthew` | `Joanna` | `Amy` | `Emma` | `Brian` | `Arthur` | `Aria` | `Ayanda`, language: string, language_code: string, engine: `generative` | `long-form` | `standard` | `neural`, credentials: object }, endpoint: { url: string, headers: object } }[] **(required)** + +**AgentV1InjectUserMessage** — Send inject user message to Deepgram's Voice Agent API + + - `type` `InjectUserMessage` **(required)** — Message type identifier for injecting a user message + - `content` string **(required)** — The specific phrase or statement the agent should respond to + +**AgentV1InjectAgentMessage** — Send inject agent message to Deepgram's Voice Agent API + + - `type` `InjectAgentMessage` **(required)** — Message type identifier for injecting an agent message + - `message` string **(required)** — The statement that the agent should say + - `behavior` `default` | `queue` | `interrupt` (default: `default`) — Controls how the injection interacts with any in-progress user or agent turn. + + * `default` — The agent speaks only if neither the user nor the agent is mid-turn. If a turn is in progress, the server replies with `InjectionRefused`. + * `queue` — The message is appended after any already-queued `ConversationText` without interrupting the current agent turn or think response. If nothing is queued, the message plays immediately. + * `interrupt` — The agent immediately speaks. If the agent was already speaking, it interrupts the current speech and replaces it with the new message. If the user is speaking, the agent interrupts with the new message, but the user's continued speech triggers `UserStartedSpeaking`, which quickly interrupts the agent. + +**AgentV1SendFunctionCallResponse** — Send a function call response from the client to the server after + executing a client-side function call. This is used when the server + requests execution of a function marked with `client_side: true`. + + - `type` `FunctionCallResponse` **(required)** — Message type identifier for function call responses + - `id` string — The unique identifier for the function call. + + • **Required for client responses**: Should match the id from + the corresponding `FunctionCallRequest` + • **Optional for server responses**: Server may omit when responding + to internal function executions + - `name` string **(required)** — The name of the function being called + - `content` string **(required)** — The content or result of the function call + +**AgentV1KeepAlive** — Send keep alive to Deepgram's Voice Agent API + + - `type` `KeepAlive` **(required)** — Message type identifier + +**AgentV1UpdatePrompt** — Send a prompt update to Deepgram's Voice Agent API + + - `type` `UpdatePrompt` **(required)** — Message type identifier for prompt update request + - `prompt` string **(required)** — The new system prompt to be used by the agent + +**AgentV1ForceEndTurn** — Send a ForceEndTurn message to immediately end the current user turn + + - `type` `ForceEndTurn` **(required)** — Message type identifier for forcing the end of the current turn + +**AgentV1Media** — Send raw binary audio data to Deepgram's Voice Agent API for processing + +#### Server → Client Messages + +**AgentV1ListenUpdated** — Receive listen update from Deepgram's Voice Agent API + + - `type` `ListenUpdated` **(required)** — Message type identifier for listen update confirmation + +**AgentV1ThinkUpdated** — Receive think update from Deepgram's Voice Agent API + + - `type` `ThinkUpdated` **(required)** — Message type identifier for think update confirmation + +**AgentV1ReceiveFunctionCallResponse** — Receive a function call response from the server after the server + has executed a server-side function call internally. This occurs + when functions are marked with `client_side: false`. + + - `type` `FunctionCallResponse` **(required)** — Message type identifier for function call responses + - `id` string — The unique identifier for the function call. + + • **Required for client responses**: Should match the id from + the corresponding `FunctionCallRequest` + • **Optional for server responses**: Server may omit when responding + to internal function executions + - `name` string **(required)** — The name of the function being called + - `content` string **(required)** — The content or result of the function call + +**AgentV1PromptUpdated** — Receive prompt update from Deepgram's Voice Agent API + + - `type` `PromptUpdated` **(required)** — Message type identifier for prompt update confirmation + +**AgentV1SpeakUpdated** — Receive speak update from Deepgram's Voice Agent API + + - `type` `SpeakUpdated` **(required)** — Message type identifier for speak update confirmation + +**AgentV1InjectionRefused** — Receive injection refused message from Deepgram's Voice Agent API + + - `type` `InjectionRefused` **(required)** — Message type identifier for injection refused + - `message` string **(required)** — Details about why the injection was refused + +**AgentV1Welcome** — Receive welcome message from Deepgram's Voice Agent API + + - `type` `Welcome` **(required)** — Message type identifier for welcome message + - `request_id` string **(required)** — Unique identifier for the request + +**AgentV1SettingsApplied** — Receive settings applied message from Deepgram's Voice Agent API + + - `type` `SettingsApplied` **(required)** — Message type identifier for settings applied confirmation + +**AgentV1ConversationText** — Receive conversation text from Deepgram's Voice Agent API + + - `type` `ConversationText` **(required)** — Message type identifier for conversation text + - `role` `user` | `assistant` **(required)** — Identifies who spoke the statement + - `content` string **(required)** — The actual statement that was spoken + - `languages_hinted` string[] — The language hints that were active at the time of the turn. Only present on user-role messages when the listen model is flux-general-multi. + - `languages` string[] — Languages detected in the user's speech, sorted by word count (descending). Only present on user-role messages when the listen model is flux-general-multi. + +**AgentV1UserStartedSpeaking** — Receive user started speaking message from Deepgram's Voice Agent API + + - `type` `UserStartedSpeaking` **(required)** — Message type identifier indicating that the user has begun speaking + +**AgentV1AgentThinking** — Receive agent thinking message from Deepgram's Voice Agent API + + - `type` `AgentThinking` **(required)** — Message type identifier for agent thinking + - `content` string **(required)** — The text of the agent's thought process + +**AgentV1LatencyReport** — Receive a latency report from Deepgram's Voice Agent API + + - `type` `LatencyReport` **(required)** — Message type identifier for the latency report + - `stt_latency` string — Speech-to-text: time from audio received to transcript produced, in seconds + - `ttt_token_latency` string — Time to first token of any type (text, tool call, or thinking), in seconds + - `ttt_text_latency` string — Time to first text token from the LLM, in seconds + - `ttt_tool_latency` string — Time to first tool-call token from the LLM, in seconds + - `ttt_thinking_latency` string — Time to first thinking token from the LLM, in seconds + - `tts_latency` string — Text-to-speech: time from first text token to first audio byte, in seconds + - `total_latency` string — End-to-end: time from user utterance end to first audio byte, in seconds + +**AgentV1FunctionCallRequest** — Receive function call request from Deepgram's Voice Agent API + + - `type` `FunctionCallRequest` **(required)** — Message type identifier for function call requests + - `functions` { id: string, name: string, arguments: string, client_side: boolean, thought_signature: string }[] **(required)** — Array of functions to be called + +**AgentV1FunctionCallCancelled** — Receive notice that a function call you already received was cancelled because the user started speaking again + + - `type` `FunctionCallCancelled` **(required)** — Message type identifier for cancelled function calls + - `functions` { id: string, name: string }[] **(required)** — The function calls that are no longer valid + +**AgentV1AgentStartedSpeaking** — Receive agent started speaking message from Deepgram's Voice Agent API + + - `type` `AgentStartedSpeaking` **(required)** — Message type identifier for agent started speaking + - `total_latency` string **(required)** — Seconds from receiving the user's utterance to producing the agent's reply + - `tts_latency` string **(required)** — The portion of total latency attributable to text-to-speech + - `ttt_latency` string **(required)** — The portion of total latency attributable to text-to-text (usually an LLM) + +**AgentV1AgentAudioDone** — Receive agent audio done message from Deepgram's Voice Agent API + + - `type` `AgentAudioDone` **(required)** — Message type identifier indicating the agent has finished sending audio + +**AgentV1Error** — Receive error response from Deepgram's Voice Agent API + + - `type` `Error` **(required)** — Message type identifier for error responses + - `description` string **(required)** — A description of what went wrong + - `code` string **(required)** — Error code identifying the type of error + +**AgentV1Warning** — Receive warning messages from Deepgram's Voice Agent API + + - `type` `Warning` **(required)** — Message type identifier for warnings + - `description` string **(required)** — Description of the warning + - `code` string **(required)** — Warning code identifier + +**AgentV1History** — Receive a conversation history message from Deepgram's Voice Agent API. Each message is either a conversation text (with role and content) or a function call record (with function_calls array). + +**AgentV1Audio** — Receive raw binary audio data generated by Deepgram's Voice Agent API diff --git a/.agents/skills/api/references/auth.md b/.agents/skills/api/references/auth.md new file mode 100644 index 0000000..5fcb8c8 --- /dev/null +++ b/.agents/skills/api/references/auth.md @@ -0,0 +1,43 @@ +# Deepgram Auth API + +Authentication — manage API keys and temporary tokens. + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### POST `/v1/auth/grant` + +Token-based Authentication + +Generates a temporary JSON Web Token (JWT) with a 30-second (by default) TTL and usage::write permission for core voice APIs, requiring an API key with Member or higher authorization. Tokens created with this endpoint will not work with the Manage APIs. + +#### Request Body + +**application/json** + +- `ttl_seconds` number (range: `1` to `3600`) — Time to live in seconds for the token. Defaults to 30 seconds. + +#### Responses + +**200**: Grant response +**400**: Invalid Request diff --git a/.agents/skills/api/references/listen.md b/.agents/skills/api/references/listen.md new file mode 100644 index 0000000..a283621 --- /dev/null +++ b/.agents/skills/api/references/listen.md @@ -0,0 +1,338 @@ +# Deepgram Listen API + +Speech-to-text transcription — convert audio and video into text. + +## Documentation + +- [Speech-to-Text Getting Started](https://developers.deepgram.com/docs/stt/getting-started) +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### POST `/v1/listen` + +Transcribe and analyze pre-recorded audio and video + +Transcribe audio and video using Deepgram's speech-to-text REST API + +#### Query Parameters + +- `callback` string — URL to which we'll make the callback request +- `callback_method` `POST` | `PUT` (default: `POST`) — HTTP method by which the callback request will be made +- `extra` string | string[] — Arbitrary key-value pairs that are attached to the API response for usage in downstream processing +- `sentiment` boolean (default: `false`) — Recognizes the sentiment throughout a transcript or text +- `summarize` `v2` | boolean — Summarize content. For Listen API, supports string version option. For Read API, accepts boolean only. +- `tag` string | string[] — Label your requests for the purpose of identification during usage reporting +- `topics` boolean (default: `false`) — Detect topics throughout a transcript or text +- `custom_topic` string | string[] — Custom topics you want the model to detect within your input audio or text if present Submit up to `100`. +- `custom_topic_mode` `extended` | `strict` (default: `extended`) — Sets how the model will interpret strings submitted to the `custom_topic` param. When `strict`, the model will only return topics submitted using the `custom_topic` param. When `extended`, the model will return its own detected topics in addition to those submitted using the `custom_topic` param +- `intents` boolean (default: `false`) — Recognizes speaker intent throughout a transcript or text +- `custom_intent` string | string[] — Custom intents you want the model to detect within your input audio if present +- `custom_intent_mode` `extended` | `strict` (default: `extended`) — Sets how the model will interpret intents submitted to the `custom_intent` param. When `strict`, the model will only return intents submitted using the `custom_intent` param. When `extended`, the model will return its own detected intents in the `custom_intent` param. +- `detect_entities` boolean (default: `false`) — Identifies and extracts key entities from content in submitted audio +- `detect_language` boolean | string[] — Identifies the dominant language spoken in submitted audio +- `diarize` boolean (default: `false`) — Deprecated: use `diarize_model` instead. Recognize speaker changes. Each word in the transcript will be assigned a speaker number starting at 0. +- `diarize_model` `latest` | `v1` | `v2` — Select and enable a specific diarization model version. Specifying this parameter enables diarization and selects the model — you do not need to also set the deprecated `diarize=true` parameter. For batch, supported values are `latest` (currently v2), `v1`, and `v2`. For streaming, supported values are `latest` (currently v1) and `v1`; `v2` returns a validation error on streaming requests. +- `dictation` boolean (default: `false`) — Dictation mode for controlling formatting with dictated speech +- `encoding` `linear16` | `flac` | `mulaw` | `amr-nb` | `amr-wb` | `opus` | `speex` | `g729` — Specify the expected encoding of your submitted audio +- `filler_words` boolean (default: `false`) — Filler Words can help transcribe interruptions in your audio, like "uh" and "um" +- `keyterm` string[] — Key term prompting improves recognition of specialized terminology and brands. Only compatible with Nova-3. + + `keyterm` accepts plain terms only. Unlike the legacy `keywords` feature, it does not support weights or intensifiers. Appending one (for example, `keyterm=term:0.15`) is not rejected—the weight is silently ignored and the entire value is treated as a literal keyterm. + + To boost multiple separate keyterms, repeat the `keyterm` parameter (for example, `keyterm=term1&keyterm=term2`). To boost one multi-word phrase as a single keyterm, join the words with `%20` or `+` (for example, `keyterm=customer%20service`). Do not separate keyterms with commas, semicolons, or line breaks. +- `keywords` string | string[] — Keywords can boost or suppress specialized terminology and brands. `keywords` is not supported with Nova-3 models; use `keyterm` instead. +- `language` string (default: `en`) — The [BCP-47 language tag](https://tools.ietf.org/html/bcp47) that hints at the primary spoken language. Depending on the Model and API endpoint you choose only certain languages are available +- `measurements` boolean (default: `false`) — Spoken measurements will be converted to their corresponding abbreviations +- `model` `nova-3` | `nova-3-general` | `nova-3-medical` | `nova-2` | `nova-2-general` | `nova-2-meeting` | `nova-2-finance` | `nova-2-conversationalai` | `nova-2-voicemail` | `nova-2-video` | `nova-2-medical` | `nova-2-drivethru` | `nova-2-automotive` | `nova` | `nova-general` | `nova-phonecall` | `nova-medical` | `enhanced` | `enhanced-general` | `enhanced-meeting` | `enhanced-phonecall` | `enhanced-finance` | `base` | `meeting` | `phonecall` | `finance` | `conversationalai` | `voicemail` | `video` | string (default: `base-general`) — AI model used to process submitted audio +- `multichannel` boolean (default: `false`) — Transcribe each audio channel independently +- `numerals` boolean (default: `false`) — Numerals converts numbers from written format to numerical format +- `paragraphs` boolean (default: `false`) — Splits audio into paragraphs to improve transcript readability +- `profanity_filter` boolean (default: `false`) — Profanity Filter looks for recognized profanity and converts it to the nearest recognized non-profane word or removes it from the transcript completely +- `punctuate` boolean (default: `false`) — Add punctuation and capitalization to the transcript +- `redact` string | `pci` | `pii` | `numbers`[] (default: `false`) — Redaction removes sensitive information from your transcripts +- `replace` string | string[] — Search for terms or phrases in submitted audio and replaces them +- `search` string | string[] — Search for terms or phrases in submitted audio +- `smart_format` boolean (default: `false`) — Apply formatting to transcript output. When set to true, additional formatting will be applied to transcripts to improve readability +- `utterances` boolean (default: `false`) — Segments speech into meaningful semantic units +- `utt_split` number (default: `0.8`) — Seconds to wait before detecting a pause between words in submitted audio +- `version` `latest` | string (default: `latest`) — Version of an AI model to use +- `mip_opt_out` boolean (default: `false`) — Opts out requests from the Deepgram Model Improvement Program. Refer to our Docs for pricing impacts before setting this to true. https://dpgr.am/deepgram-mip + +#### Request Body + +**application/json** + +- `url` string **(required)** + +#### Responses + +**200**: Returns either transcription results, or a request_id when using a callback. +**400**: Invalid Request + +## WebSocket API + +### WebSocket `/v1/listen` +> Server: `wss://api.deepgram.com` + +Transcribe audio and video using Deepgram's speech-to-text WebSocket + +#### Connection Parameters + +- `callback` string — URL to which we'll make the callback request +- `callback_method` `POST` | `GET` | `PUT` | `DELETE` (default: `POST`) — HTTP method by which the callback request will be made +- `channels` any (default: `1`) — Any type +- `detect_entities` `true` | `false` (default: `false`) — Identifies and extracts key entities from content in submitted audio. Entities appear in final results. When enabled, Punctuation will also be enabled by default +- `diarize` `true` | `false` (default: `false`) — Deprecated. Use `diarize_model` instead. Defaults to `false`. Recognize speaker changes. Each word in the transcript will be assigned a speaker number starting at 0 +- `diarize_model` `latest` | `v1` +- `dictation` `true` | `false` (default: `false`) — Identify and extract key entities from content in submitted audio +- `encoding` `linear16` | `linear32` | `flac` | `alaw` | `mulaw` | `amr-nb` | `amr-wb` | `opus` | `ogg-opus` | `speex` | `g729` — Specify the expected encoding of your submitted audio +- `endpointing` any (default: `10`) — Any type +- `extra` string | string[] — Arbitrary key-value pairs that are attached to the API response for usage in downstream processing +- `interim_results` `true` | `false` (default: `false`) — Specifies whether the streaming endpoint should provide ongoing transcription updates as more audio is received. When set to true, the endpoint sends continuous updates, meaning transcription results may evolve over time +- `keyterm` string[] — Key term prompting improves recognition of specialized terminology and brands. Only compatible with Nova-3. + + `keyterm` accepts plain terms only. Unlike the legacy `keywords` feature, it does not support weights or intensifiers. Appending one (for example, `keyterm=term:0.15`) is not rejected—the weight is silently ignored and the entire value is treated as a literal keyterm. + + To boost multiple separate keyterms, repeat the `keyterm` parameter (for example, `keyterm=term1&keyterm=term2`). To boost one multi-word phrase as a single keyterm, join the words with `%20` or `+` (for example, `keyterm=customer%20service`). Do not separate keyterms with commas, semicolons, or line breaks. +- `keywords` string | string[] — Keywords can boost or suppress specialized terminology and brands. `keywords` is not supported with Nova-3 models; use `keyterm` instead. +- `language` string (default: `en`) — The [BCP-47 language tag](https://tools.ietf.org/html/bcp47) that hints at the primary spoken language. Depending on the Model and API endpoint you choose only certain languages are available +- `mip_opt_out` boolean (default: `false`) — Opts out requests from the Deepgram Model Improvement Program. Refer to our Docs for pricing impacts before setting this to true. https://dpgr.am/deepgram-mip +- `model` `nova-3` | `nova-3-general` | `nova-3-medical` | `nova-2` | `nova-2-general` | `nova-2-meeting` | `nova-2-finance` | `nova-2-conversationalai` | `nova-2-voicemail` | `nova-2-video` | `nova-2-medical` | `nova-2-drivethru` | `nova-2-automotive` | `nova` | `nova-general` | `nova-phonecall` | `nova-medical` | `enhanced` | `enhanced-general` | `enhanced-meeting` | `enhanced-phonecall` | `enhanced-finance` | `base` | `meeting` | `phonecall` | `finance` | `conversationalai` | `voicemail` | `video` | `custom` — AI model to use for the transcription +- `multichannel` `true` | `false` (default: `false`) — Transcribe each audio channel independently +- `numerals` `true` | `false` (default: `false`) — Convert numbers from written format to numerical format +- `profanity_filter` `true` | `false` (default: `false`) — Profanity Filter looks for recognized profanity and converts it to the nearest recognized non-profane word or removes it from the transcript completely +- `punctuate` `true` | `false` (default: `false`) — Add punctuation and capitalization to the transcript +- `redact` `true` | `false` | `pci` | `numbers` | `aggressive_numbers` | `ssn` (default: `false`) — Redaction removes sensitive information from your transcripts +- `replace` string | string[] — Search for terms or phrases in submitted audio and replaces them +- `sample_rate` any — Any type +- `search` string | string[] — Search for terms or phrases in submitted audio +- `smart_format` `true` | `false` (default: `false`) — Apply formatting to transcript output. When set to true, additional formatting will be applied to transcripts to improve readability +- `tag` string | string[] — Label your requests for the purpose of identification during usage reporting +- `utterance_end_ms` any — Any type +- `vad_events` `true` | `false` (default: `false`) — Indicates that speech has started. You'll begin receiving Speech Started messages upon speech starting +- `version` `latest` | string (default: `latest`) — Version of an AI model to use + +#### Client → Server Messages + +**ListenV1Media** — Send audio or video data to be transcribed + +**ListenV1Finalize** — Send a Finalize message to flush the WebSocket stream + + - `type` `Finalize` | `CloseStream` | `KeepAlive` **(required)** — Message type identifier + +**ListenV1CloseStream** — Send a CloseStream message to close the WebSocket stream + + - `type` `Finalize` | `CloseStream` | `KeepAlive` **(required)** — Message type identifier + +**ListenV1KeepAlive** — Send a KeepAlive message to keep the WebSocket stream alive + + - `type` `Finalize` | `CloseStream` | `KeepAlive` **(required)** — Message type identifier + +#### Server → Client Messages + +**ListenV1Results** — Receive transcription results + + - `type` `Results` **(required)** — Message type identifier + - `channel_index` integer[] **(required)** — The index of the channel + - `duration` number **(required)** — The duration of the transcription + - `start` number **(required)** — The start time of the transcription + - `is_final` boolean — Whether the transcription is final + - `speech_final` boolean — Whether the transcription is speech final + - `channel` { alternatives: { transcript: string, confidence: number, languages: string[], words: object[] }[] } **(required)** + - `metadata` { request_id: string, model_info: { name: string, version: string, arch: string }, model_uuid: string, diarize_info: { model_uuid: string, arch: string } } **(required)** + - `from_finalize` boolean — Whether the transcription is from a finalize message + - `entities` { label: string, value: string, raw_value: string, confidence: number, start_word: integer, end_word: integer }[] — Extracted entities from the audio when detect_entities is enabled. Only present in is_final messages. Returns an empty array if no entities are detected + +**ListenV1Metadata** — Receive metadata about the transcription + + - `type` `Metadata` **(required)** — Message type identifier + - `transaction_key` string **(required)** — The transaction key + - `request_id` string **(required)** — The request ID + - `sha256` string **(required)** — The sha256 + - `created` string **(required)** — The created + - `duration` number **(required)** — The duration + - `channels` integer **(required)** — The channels + +**ListenV1UtteranceEnd** — Receive an utterance end event + + - `type` `UtteranceEnd` **(required)** — Message type identifier + - `channel` integer[] **(required)** — The channel + - `last_word_end` number **(required)** — The last word end + +**ListenV1SpeechStarted** — Receive a speech started event + + - `type` `SpeechStarted` **(required)** — Message type identifier + - `channel` integer[] **(required)** — The channel + - `timestamp` number **(required)** — The timestamp + +### WebSocket `/v2/listen` +> Server: `wss://api.deepgram.com` + +Real-time conversational speech recognition with contextual turn detection +for natural voice conversations + + +#### Connection Parameters + +- `model` `flux-general-en` | `flux-general-multi` — Defines the AI model used to process submitted audio. +- `encoding` `linear16` | `linear32` | `mulaw` | `alaw` | `opus` | `ogg-opus` — Encoding of the audio stream. Required if sending non-containerized/raw audio. If sending containerized audio, this parameter should be omitted. +- `sample_rate` any — Any type +- `eager_eot_threshold` number — End-of-turn confidence required to fire an eager end-of-turn event. When set, enables EagerEndOfTurn and TurnResumed events. Valid range: 0.3 - 0.9. +- `eot_threshold` number (default: `0.7`) — End-of-turn confidence required to finish a turn. Valid range: 0.5 - 1.0. Defaults to 0.7. Set to 1.0 to fully suppress confidence-based end-of-turn detection. `eot_timeout_ms` still ends idle turns; increase it when using ForceEndTurn for full manual turn control. +- `eot_timeout_ms` integer (default: `5000`) — A turn will be finished when this much time in milliseconds has passed after speech, regardless of EOT confidence. Defaults to 5000. +- `keyterm` string | string[] — Keyterm prompting improves recognition of specialized terminology. + + `keyterm` accepts plain terms only. Unlike the legacy `keywords` feature, + it does not support weights or intensifiers. Appending one + (for example, `keyterm=term:0.15`) is not rejected—the weight is + silently ignored and the entire value is treated as a literal keyterm. + + To boost multiple separate keyterms, repeat the `keyterm` parameter + (for example, `keyterm=term1&keyterm=term2`). To boost one multi-word + phrase as a single keyterm, join the words with `%20` or `+` + (for example, `keyterm=customer%20service`). Do not separate keyterms + with commas, semicolons, or line breaks. +- `language_hint` string | string[] — Language hints constrain and prioritize language detection for the + flux-general-multi model. Pass multiple language_hint query parameters + to specify multiple language codes. Empty values are rejected. + Only valid when model is flux-general-multi. +- `profanity_filter` `true` | `false` (default: `false`) — Profanity Filter looks for recognized profanity and converts it to the nearest recognized non-profane word or removes it from the transcript completely. +- `numerals` `true` | `false` (default: `false`) — Numerals converts numbers from written format to numerical format +- `redact` `numbers` | `aggressive_numbers` — Redaction removes sensitive information from your transcripts. On Flux, only `numbers` and `aggressive_numbers` are supported. +- `mip_opt_out` any — Any type +- `tag` any — Any type + +#### Client → Server Messages + +**ListenV2Media** — Send audio or video data to be transcribed + +**ListenV2CloseStream** — Send a CloseStream message to close the WebSocket stream + + - `type` `CloseStream` **(required)** — Message type identifier + +**ListenV2ForceEndTurn** — Send a ForceEndTurn message to immediately end the current turn + + - `type` `ForceEndTurn` **(required)** — Message type identifier + +**ListenV2Configure** — Send a Configure message to update Flux settings + + - `type` `Configure` **(required)** — Message type identifier + - `thresholds` { eager_eot_threshold: any, eot_threshold: any, eot_timeout_ms: any } — Updates each parameter, if it is supplied. If a particular threshold parameter + is not supplied, the configuration continues using the currently configured value. + - `keyterms` string | string[] — Keyterm prompting improves recognition of specialized terminology. + + `keyterm` accepts plain terms only. Unlike the legacy `keywords` feature, + it does not support weights or intensifiers. Appending one + (for example, `keyterm=term:0.15`) is not rejected—the weight is + silently ignored and the entire value is treated as a literal keyterm. + + To boost multiple separate keyterms, repeat the `keyterm` parameter + (for example, `keyterm=term1&keyterm=term2`). To boost one multi-word + phrase as a single keyterm, join the words with `%20` or `+` + (for example, `keyterm=customer%20service`). Do not separate keyterms + with commas, semicolons, or line breaks. + - `language_hints` string[] — Language hints to constrain and prioritize language detection. + Only valid when the model is flux-general-multi. If this field is not supplied, + the session will continue to use the currently configured value. + - `numerals` boolean (default: `false`) — Numerals converts numbers from written format to numerical format. Applies to turns transcribed after the update. + +#### Server → Client Messages + +**ListenV2Connected** — Receive a connected message + + - `type` `Connected` **(required)** — Message type identifier + - `request_id` string **(required)** — The unique identifier of the request + - `sequence_id` integer **(required)** — Starts at `0` and increments for each message the server sends + to the client. This includes messages of other types, like + `TurnInfo` messages. + +**ListenV2TurnInfo** — Receive a turn info message + + - `type` `TurnInfo` **(required)** + - `request_id` string **(required)** — The unique identifier of the request + - `sequence_id` integer **(required)** — Starts at `0` and increments for each message the server sends to the client. This includes messages of other types, like `Connected` messages. + - `event` `Update` | `StartOfTurn` | `EagerEndOfTurn` | `TurnResumed` | `EndOfTurn` **(required)** — The type of event being reported. + + - **Update** - Additional audio has been transcribed, but the turn state hasn't changed + - **StartOfTurn** - The user has begun speaking for the first time in the turn + - **EagerEndOfTurn** - The system has moderate confidence that the user has finished speaking for the turn. This is an opportunity to begin preparing an agent reply + - **TurnResumed** - The system detected that speech had ended and therefore sent an **EagerEndOfTurn** event, but speech is actually continuing for this turn + - **EndOfTurn** - The user has finished speaking for the turn + - `turn_index` integer **(required)** (minimum: `0`) — The index of the current turn + - `audio_window_start` string **(required)** — Start time in seconds of the audio range that was transcribed + - `audio_window_end` string **(required)** — End time in seconds of the audio range that was transcribed + - `transcript` string **(required)** — Text that was said over the course of the current turn + - `words` { word: string, confidence: string, start: number, end: number }[] **(required)** — The words in the `transcript` + - `end_of_turn_confidence` string **(required)** — Confidence that no more speech is coming in this turn + - `trigger` string — The cause of the turn ending. Present on every `EndOfTurn` event and only there. + + - **model** - the turn ended by Flux's native end-of-turn detection + + - **manual** - the turn ended because a `ForceEndTurn` message was sent + + - **timeout** - the turn ended because `eot_timeout_ms` elapsed + + This is an open enum. New values may be added over time, so clients must tolerate values they do not recognize. + - `languages` string[] — Detected languages sorted by descending frequency in the + transcript. Only present when the flux-general-multi model + detects languages in the audio. + - `languages_hinted` string[] — The language hints that were supplied for this turn. Only + present when language hints are configured. + +**ListenV2ConfigureSuccess** — Sent when a `Configure` message was successfully applied. Returns the current, up-to-date values that were applied. + + - `type` `ConfigureSuccess` **(required)** — Message type identifier + - `request_id` string **(required)** — The unique identifier of the request + - `thresholds` { eager_eot_threshold: any, eot_threshold: any, eot_timeout_ms: any } **(required)** — Updates each parameter, if it is supplied. If a particular threshold parameter + is not supplied, the configuration continues using the currently configured value. + - `keyterms` string | string[] **(required)** — Keyterm prompting improves recognition of specialized terminology. + + `keyterm` accepts plain terms only. Unlike the legacy `keywords` feature, + it does not support weights or intensifiers. Appending one + (for example, `keyterm=term:0.15`) is not rejected—the weight is + silently ignored and the entire value is treated as a literal keyterm. + + To boost multiple separate keyterms, repeat the `keyterm` parameter + (for example, `keyterm=term1&keyterm=term2`). To boost one multi-word + phrase as a single keyterm, join the words with `%20` or `+` + (for example, `keyterm=customer%20service`). Do not separate keyterms + with commas, semicolons, or line breaks. + - `language_hints` string[] — The currently active language hints. Only applicable to the flux-general-multi model. + - `sequence_id` integer **(required)** — Starts at `0` and increments for each message the server sends + to the client. This includes messages of other types, like + `TurnInfo` messages. + +**ListenV2ConfigureFailure** — Indicates that a Configure message was rejected + + - `type` `ConfigureFailure` **(required)** — Message type identifier + - `request_id` string **(required)** — The unique identifier of the request + - `sequence_id` integer **(required)** — Starts at `0` and increments for each message the server sends + to the client. This includes messages of other types, like + `TurnInfo` messages. + +**ListenV2FatalError** — Receive a fatal error message + + - `type` `Error` **(required)** — Message type identifier + - `sequence_id` integer **(required)** — Starts at `0` and increments for each message the server sends + to the client. This includes messages of other types, like + `Connected` messages. + - `code` string **(required)** — A string code describing the error, e.g. `INTERNAL_SERVER_ERROR` + - `description` string **(required)** — Prose description of the error diff --git a/.agents/skills/api/references/models.md b/.agents/skills/api/references/models.md new file mode 100644 index 0000000..0442c08 --- /dev/null +++ b/.agents/skills/api/references/models.md @@ -0,0 +1,78 @@ +# Deepgram Models API + +Model management — list and query available models. + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### GET `/v1/projects/{project_id}/models` + +List Project Models + +Returns metadata on all the latest models that a specific project has access to, including non-public models + +#### Query Parameters + +- `include_outdated` boolean — returns non-latest versions of models + +#### Responses + +**200**: A list of models +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/models/{model_id}` + +Get a Project Model + +Returns metadata for a specific model + +#### Responses + +**200**: A model object that can be either STT or TTS +**400**: Invalid Request + +### GET `/v1/models` + +List Models + +Returns metadata on all the latest public models. To retrieve custom models, use Get Project Models. + +#### Query Parameters + +- `include_outdated` boolean — returns non-latest versions of models + +#### Responses + +**200**: A list of models +**400**: Invalid Request + +### GET `/v1/models/{model_id}` + +Get a specific Model + +Returns metadata for a specific public model + +#### Responses + +**200**: A model object that can be either STT or TTS +**400**: Invalid Request diff --git a/.agents/skills/api/references/projects.md b/.agents/skills/api/references/projects.md new file mode 100644 index 0000000..b1288e1 --- /dev/null +++ b/.agents/skills/api/references/projects.md @@ -0,0 +1,476 @@ +# Deepgram Projects API + +Project management — manage projects, keys, members, and usage. + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### GET `/v1/projects` + +List Projects + +Retrieves basic information about the projects associated with the API key + +#### Responses + +**200**: A list of projects +**400**: Invalid Request + +### GET `/v1/projects/{project_id}` + +Get a Project + +Retrieves information about the specified project + +#### Query Parameters + +- `limit` number (default: `10`) — Number of results to return per page. Default 10. Range [1,1000] +- `page` number — Navigate and return the results to retrieve specific portions of information of the response + +#### Responses + +**200**: A project +**400**: Invalid Request + +### PATCH `/v1/projects/{project_id}` + +Update a Project + +Updates the name or other properties of an existing project + +#### Request Body + +**application/json** + +- `name` string — The name of the project + +#### Responses + +**200**: A project +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}` + +Delete a Project + +Deletes the specified project + +#### Responses + +**200**: A project +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/leave` + +Leave a Project + +Removes the authenticated account from the specific project + +#### Responses + +**200**: Successfully removed account from project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/keys` + +List Project Keys + +Retrieves all API keys associated with the specified project + +#### Query Parameters + +- `status` `active` | `expired` — Only return keys with a specific status + +#### Responses + +**200**: A list of API keys +**400**: Invalid Request + +### POST `/v1/projects/{project_id}/keys` + +Create a Project Key + +Creates a new API key with specified settings for the project + +#### Request Body + +**application/json** + +#### Responses + +**200**: API key created successfully +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/keys/{key_id}` + +Get a Project Key + +Retrieves information about a specified API key + +#### Responses + +**200**: A specific API key +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/keys/{key_id}` + +Delete a Project Key + +Deletes an API key for a specific project + +#### Responses + +**200**: API key deleted +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/members` + +List Project Members + +Retrieves a list of members for a given project + +#### Responses + +**200**: A list of members for a given project +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/members/{member_id}` + +Delete a Project Member + +Removes a member from the project using their unique member ID + +#### Responses + +**200**: Delete the specific member from the project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/members/{member_id}/scopes` + +List Project Member Scopes + +Retrieves a list of scopes for a specific member + +#### Responses + +**200**: A list of scopes for a specific member +**400**: Invalid Request + +### PUT `/v1/projects/{project_id}/members/{member_id}/scopes` + +Update Project Member Scopes + +Updates the scopes for a specific member + +#### Request Body + +**application/json** + +- `scope` string **(required)** — A scope to update + +#### Responses + +**200**: Updated the scopes for a specific member +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/invites` + +List Project Invites + +Generates a list of invites for a specific project + +#### Responses + +**200**: A list of invites for a specific project +**400**: Invalid Request + +### POST `/v1/projects/{project_id}/invites` + +Create a Project Invite + +Generates an invite for a specific project + +#### Request Body + +**application/json** + +- `email` string **(required)** — The email address of the invitee +- `scope` string **(required)** — The scope of the invitee + +#### Responses + +**200**: The invite was successfully generated +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/invites/{email}` + +Delete a Project Invite + +Deletes an invite for a specific project + +#### Responses + +**200**: The invite was successfully deleted +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/requests` + +List Project Requests + +Generates a list of requests for a specific project + +#### Query Parameters + +- `start` string — Start date of the requested date range. Formats accepted are YYYY-MM-DD, YYYY-MM-DDTHH:MM:SS, or YYYY-MM-DDTHH:MM:SS+HH:MM +- `end` string — End date of the requested date range. Formats accepted are YYYY-MM-DD, YYYY-MM-DDTHH:MM:SS, or YYYY-MM-DDTHH:MM:SS+HH:MM +- `limit` number (default: `10`) — Number of results to return per page. Default 10. Range [1,1000] +- `page` number — Navigate and return the results to retrieve specific portions of information of the response +- `accessor` string — Filter for requests where a specific accessor was used +- `request_id` string — Filter for a specific request id +- `deployment` `hosted` | `beta` | `self-hosted` — Filter for requests where a specific deployment was used +- `endpoint` `listen` | `read` | `speak` | `agent` — Filter for requests where a specific endpoint was used +- `method` `sync` | `async` | `streaming` — Filter for requests where a specific method was used +- `status` `succeeded` | `failed` — Filter for requests that succeeded (status code < 300) or failed (status code >=400) + +#### Responses + +**200**: A list of requests for a specific project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/requests/{request_id}` + +Get a Project Request + +Retrieves a specific request for a specific project + +#### Responses + +**200**: A specific request for a specific project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/usage` + +Get Project Usage + +Retrieves the usage for a specific project. Use Get Project Usage Breakdown for a more comprehensive usage summary. + +#### Query Parameters + +- `start` string — Start date of the requested date range. Format accepted is YYYY-MM-DD +- `end` string — End date of the requested date range. Format accepted is YYYY-MM-DD +- `accessor` string — Filter for requests where a specific accessor was used +- `alternatives` boolean — Filter for requests where alternatives were used +- `callback_method` boolean — Filter for requests where callback method was used +- `callback` boolean — Filter for requests where callback was used +- `channels` boolean — Filter for requests where channels were used +- `custom_intent_mode` boolean — Filter for requests where custom intent mode was used +- `custom_intent` boolean — Filter for requests where custom intent was used +- `custom_topic_mode` boolean — Filter for requests where custom topic mode was used +- `custom_topic` boolean — Filter for requests where custom topic was used +- `deployment` `hosted` | `beta` | `self-hosted` — Filter for requests where a specific deployment was used +- `detect_entities` boolean — Filter for requests where detect entities was used +- `detect_language` boolean — Filter for requests where detect language was used +- `diarize` boolean — Filter for requests where diarize was used +- `dictation` boolean — Filter for requests where dictation was used +- `encoding` boolean — Filter for requests where encoding was used +- `endpoint` `listen` | `read` | `speak` | `agent` — Filter for requests where a specific endpoint was used +- `extra` boolean — Filter for requests where extra was used +- `filler_words` boolean — Filter for requests where filler words was used +- `intents` boolean — Filter for requests where intents was used +- `keyterm` boolean — Filter for requests where keyterm was used +- `keywords` boolean — Filter for requests where keywords was used +- `language` boolean — Filter for requests where language was used +- `measurements` boolean — Filter for requests where measurements were used +- `method` `sync` | `async` | `streaming` — Filter for requests where a specific method was used +- `model` string — Filter for requests where a specific model uuid was used +- `multichannel` boolean — Filter for requests where multichannel was used +- `numerals` boolean — Filter for requests where numerals were used +- `paragraphs` boolean — Filter for requests where paragraphs were used +- `profanity_filter` boolean — Filter for requests where profanity filter was used +- `punctuate` boolean — Filter for requests where punctuate was used +- `redact` boolean — Filter for requests where redact was used +- `replace` boolean — Filter for requests where replace was used +- `sample_rate` boolean — Filter for requests where sample rate was used +- `search` boolean — Filter for requests where search was used +- `sentiment` boolean — Filter for requests where sentiment was used +- `smart_format` boolean — Filter for requests where smart format was used +- `summarize` boolean — Filter for requests where summarize was used +- `tag` string — Filter for requests where a specific tag was used +- `topics` boolean — Filter for requests where topics was used +- `utt_split` boolean — Filter for requests where utt split was used +- `utterances` boolean — Filter for requests where utterances was used +- `version` boolean — Filter for requests where version was used + +#### Responses + +**200**: A specific request for a specific project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/usage/fields` + +List Project Usage Fields + +Lists the features, models, tags, languages, and processing method used for requests in the specified project + +#### Query Parameters + +- `start` string — Start date of the requested date range. Format accepted is YYYY-MM-DD +- `end` string — End date of the requested date range. Format accepted is YYYY-MM-DD + +#### Responses + +**200**: A list of fields for a specific project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/usage/breakdown` + +Get Project Usage Breakdown + +Retrieves the usage breakdown for a specific project, with various filter options by API feature or by groupings. Setting a feature (e.g. diarize) to true includes requests that used that feature, while false excludes requests that used it. Multiple true filters are combined with OR logic, while false filters use AND logic. + +#### Query Parameters + +- `start` string — Start date of the requested date range. Format accepted is YYYY-MM-DD +- `end` string — End date of the requested date range. Format accepted is YYYY-MM-DD +- `grouping` `accessor` | `endpoint` | `feature_set` | `models` | `method` | `tags` | `deployment` — Common usage grouping parameters +- `accessor` string — Filter for requests where a specific accessor was used +- `alternatives` boolean — Filter for requests where alternatives were used +- `callback_method` boolean — Filter for requests where callback method was used +- `callback` boolean — Filter for requests where callback was used +- `channels` boolean — Filter for requests where channels were used +- `custom_intent_mode` boolean — Filter for requests where custom intent mode was used +- `custom_intent` boolean — Filter for requests where custom intent was used +- `custom_topic_mode` boolean — Filter for requests where custom topic mode was used +- `custom_topic` boolean — Filter for requests where custom topic was used +- `deployment` `hosted` | `beta` | `self-hosted` — Filter for requests where a specific deployment was used +- `detect_entities` boolean — Filter for requests where detect entities was used +- `detect_language` boolean — Filter for requests where detect language was used +- `diarize` boolean — Filter for requests where diarize was used +- `dictation` boolean — Filter for requests where dictation was used +- `encoding` boolean — Filter for requests where encoding was used +- `endpoint` `listen` | `read` | `speak` | `agent` — Filter for requests where a specific endpoint was used +- `extra` boolean — Filter for requests where extra was used +- `filler_words` boolean — Filter for requests where filler words was used +- `intents` boolean — Filter for requests where intents was used +- `keyterm` boolean — Filter for requests where keyterm was used +- `keywords` boolean — Filter for requests where keywords was used +- `language` boolean — Filter for requests where language was used +- `measurements` boolean — Filter for requests where measurements were used +- `method` `sync` | `async` | `streaming` — Filter for requests where a specific method was used +- `model` string — Filter for requests where a specific model uuid was used +- `multichannel` boolean — Filter for requests where multichannel was used +- `numerals` boolean — Filter for requests where numerals were used +- `paragraphs` boolean — Filter for requests where paragraphs were used +- `profanity_filter` boolean — Filter for requests where profanity filter was used +- `punctuate` boolean — Filter for requests where punctuate was used +- `redact` boolean — Filter for requests where redact was used +- `replace` boolean — Filter for requests where replace was used +- `sample_rate` boolean — Filter for requests where sample rate was used +- `search` boolean — Filter for requests where search was used +- `sentiment` boolean — Filter for requests where sentiment was used +- `smart_format` boolean — Filter for requests where smart format was used +- `summarize` boolean — Filter for requests where summarize was used +- `tag` string — Filter for requests where a specific tag was used +- `topics` boolean — Filter for requests where topics was used +- `utt_split` boolean — Filter for requests where utt split was used +- `utterances` boolean — Filter for requests where utterances was used +- `version` boolean — Filter for requests where version was used + +#### Responses + +**200**: Usage breakdown response +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/balances` + +Get Project Balances + +Generates a list of outstanding balances for the specified project + +#### Responses + +**200**: A list of outstanding balances +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/balances/{balance_id}` + +Get a Project Balance + +Retrieves details about the specified balance + +#### Responses + +**200**: A specific balance +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/billing/breakdown` + +Get Project Billing Breakdown + +Retrieves the billing summary for a specific project, with various filter options or by grouping options. + +#### Query Parameters + +- `start` string — Start date of the requested date range. Format accepted is YYYY-MM-DD +- `end` string — End date of the requested date range. Format accepted is YYYY-MM-DD +- `accessor` string — Filter for requests where a specific accessor was used +- `deployment` `hosted` | `beta` | `self-hosted` — Filter for requests where a specific deployment was used +- `tag` string — Filter for requests where a specific tag was used +- `line_item` string — Filter requests by line item (e.g. streaming::nova-3) +- `grouping` `accessor` | `deployment` | `line_item` | `tags`[] — Group billing breakdown by one or more dimensions (accessor, deployment, line_item, tags) + +#### Responses + +**200**: Billing breakdown response +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/billing/fields` + +List Project Billing Fields + +Lists the accessors, deployment types, tags, and line items used for billing data in the specified time period. Use this endpoint if you want to filter your results from the Billing Breakdown endpoint and want to know what filters are available. + +#### Query Parameters + +- `start` string — Start date of the requested date range. Format accepted is YYYY-MM-DD +- `end` string — End date of the requested date range. Format accepted is YYYY-MM-DD + +#### Responses + +**200**: A list of billing fields for a specific project +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/purchases` + +List Project Purchases + +Returns the original purchased amount on an order transaction + +#### Query Parameters + +- `limit` number (default: `10`) — Number of results to return per page. Default 10. Range [1,1000] + +#### Responses + +**200**: A list of purchases for a specific project +**400**: Invalid Request diff --git a/.agents/skills/api/references/read.md b/.agents/skills/api/references/read.md new file mode 100644 index 0000000..8376168 --- /dev/null +++ b/.agents/skills/api/references/read.md @@ -0,0 +1,57 @@ +# Deepgram Read API + +Text analysis — analyze and understand text content. + +## Documentation + +- [Text and Audio Intelligence](https://developers.deepgram.com/docs/audio-intelligence) +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### POST `/v1/read` + +Analyze text content + +Analyze text content using Deepgrams text analysis API + +#### Query Parameters + +- `callback` string — URL to which we'll make the callback request +- `callback_method` `POST` | `PUT` (default: `POST`) — HTTP method by which the callback request will be made +- `sentiment` boolean (default: `false`) — Recognizes the sentiment throughout a transcript or text +- `summarize` `v2` | boolean — Summarize content. For Listen API, supports string version option. For Read API, accepts boolean only. +- `tag` string | string[] — Label your requests for the purpose of identification during usage reporting +- `topics` boolean (default: `false`) — Detect topics throughout a transcript or text +- `custom_topic` string | string[] — Custom topics you want the model to detect within your input audio or text if present Submit up to `100`. +- `custom_topic_mode` `extended` | `strict` (default: `extended`) — Sets how the model will interpret strings submitted to the `custom_topic` param. When `strict`, the model will only return topics submitted using the `custom_topic` param. When `extended`, the model will return its own detected topics in addition to those submitted using the `custom_topic` param +- `intents` boolean (default: `false`) — Recognizes speaker intent throughout a transcript or text +- `custom_intent` string | string[] — Custom intents you want the model to detect within your input audio if present +- `custom_intent_mode` `extended` | `strict` (default: `extended`) — Sets how the model will interpret intents submitted to the `custom_intent` param. When `strict`, the model will only return intents submitted using the `custom_intent` param. When `extended`, the model will return its own detected intents in the `custom_intent` param. +- `language` string (default: `en`) — The [BCP-47 language tag](https://tools.ietf.org/html/bcp47) that hints at the primary spoken language. Depending on the Model and API endpoint you choose only certain languages are available + +#### Request Body + +**application/json** + +#### Responses + +**200**: Successful text analysis +**400**: Invalid Request diff --git a/.agents/skills/api/references/self-hosted.md b/.agents/skills/api/references/self-hosted.md new file mode 100644 index 0000000..6ad8a16 --- /dev/null +++ b/.agents/skills/api/references/self-hosted.md @@ -0,0 +1,82 @@ +# Deepgram Self-Hosted API + +Self-hosted deployments — manage the distribution credentials used to pull Deepgram container images. + +## Documentation + +- [Self-Hosted Deployments](https://developers.deepgram.com/docs/self-hosted-introduction) +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### GET `/v1/projects/{project_id}/self-hosted/distribution/credentials` + +List Project Self-Hosted Distribution Credentials + +Lists sets of distribution credentials for the specified project + +#### Responses + +**200**: A list of distribution credentials for a specific project +**400**: Invalid Request + +### POST `/v1/projects/{project_id}/self-hosted/distribution/credentials` + +Create a Project Self-Hosted Distribution Credential + +Creates a set of distribution credentials for the specified project + +#### Query Parameters + +- `scopes` `self-hosted:products` | `self-hosted:product:api` | `self-hosted:product:engine` | `self-hosted:product:license-proxy` | `self-hosted:product:dgtools` | `self-hosted:product:billing` | `self-hosted:product:hotpepper` | `self-hosted:product:metrics-server`[] (default: `self-hosted:products`) — List of permission scopes for the credentials +- `provider` `quay` (default: `quay`) — The provider of the distribution service + +#### Request Body + +**application/json** + +- `comment` string — Optional comment about the credentials + +#### Responses + +**200**: Single distribution credential +**400**: Invalid Request + +### GET `/v1/projects/{project_id}/self-hosted/distribution/credentials/{distribution_credentials_id}` + +Get a Project Self-Hosted Distribution Credential + +Returns a set of distribution credentials for the specified project + +#### Responses + +**200**: Single distribution credential +**400**: Invalid Request + +### DELETE `/v1/projects/{project_id}/self-hosted/distribution/credentials/{distribution_credentials_id}` + +Delete a Project Self-Hosted Distribution Credential + +Deletes a set of distribution credentials for the specified project + +#### Responses + +**200**: Single distribution credential +**400**: Invalid Request diff --git a/.agents/skills/api/references/speak.md b/.agents/skills/api/references/speak.md new file mode 100644 index 0000000..7e53952 --- /dev/null +++ b/.agents/skills/api/references/speak.md @@ -0,0 +1,274 @@ +# Deepgram Speak API + +Text-to-speech synthesis — convert text into natural-sounding audio. + +## Documentation + +- [Text-to-Speech Docs](https://developers.deepgram.com/docs/tts-rest) +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## Authentication + +All API requests require authentication. Two methods are supported: + +### ApiKeyAuth + +Use `Authorization: Token ` +Example: `Authorization: Token 12345abcdef` + + +### JwtAuth + +Use `Authorization: Bearer ` +Example: `Authorization: Bearer eyJhbGciOiJ...` + + + +## REST API + +### POST `/v1/speak` + +Text to Speech transformation + +Convert text into natural-sounding speech using Deepgram's TTS REST API + +#### Query Parameters + +- `callback` string — URL to which we'll make the callback request +- `callback_method` `POST` | `PUT` (default: `POST`) — HTTP method by which the callback request will be made +- `mip_opt_out` boolean (default: `false`) — Opts out requests from the Deepgram Model Improvement Program. Refer to our Docs for pricing impacts before setting this to true. https://dpgr.am/deepgram-mip +- `tag` string | string[] — Label your requests for the purpose of identification during usage reporting +- `bit_rate` `32000` | `48000` | number | number (default: `48000`) — The bitrate of the audio in bits per second. Choose from predefined ranges or specific values based on the encoding type. +- `container` `none` | `wav` | `wav` | `wav` | `ogg` (default: `wav`) — Container specifies the file format wrapper for the output audio. The available options depend on the encoding type. +- `encoding` `linear16` | `flac` | `mulaw` | `alaw` | `mp3` | `opus` | `aac` (default: `mp3`) — Encoding allows you to specify the expected encoding of your audio output +- `model` `aura-angus-en` | `aura-arcas-en` | `aura-asteria-en` | `aura-athena-en` | `aura-helios-en` | `aura-hera-en` | `aura-luna-en` | `aura-orion-en` | `aura-orpheus-en` | `aura-perseus-en` | `aura-stella-en` | `aura-zeus-en` | `aura-2-amalthea-en` | `aura-2-andromeda-en` | `aura-2-apollo-en` | `aura-2-arcas-en` | `aura-2-aries-en` | `aura-2-asteria-en` | `aura-2-athena-en` | `aura-2-atlas-en` | `aura-2-aurora-en` | `aura-2-callista-en` | `aura-2-cora-en` | `aura-2-cordelia-en` | `aura-2-delia-en` | `aura-2-draco-en` | `aura-2-electra-en` | `aura-2-harmonia-en` | `aura-2-helena-en` | `aura-2-hera-en` | `aura-2-hermes-en` | `aura-2-hyperion-en` | `aura-2-iris-en` | `aura-2-janus-en` | `aura-2-juno-en` | `aura-2-jupiter-en` | `aura-2-luna-en` | `aura-2-mars-en` | `aura-2-minerva-en` | `aura-2-neptune-en` | `aura-2-odysseus-en` | `aura-2-ophelia-en` | `aura-2-orion-en` | `aura-2-orpheus-en` | `aura-2-pandora-en` | `aura-2-phoebe-en` | `aura-2-pluto-en` | `aura-2-saturn-en` | `aura-2-selene-en` | `aura-2-thalia-en` | `aura-2-theia-en` | `aura-2-vesta-en` | `aura-2-zeus-en` | `aura-2-agustina-es` | `aura-2-alvaro-es` | `aura-2-antonia-es` | `aura-2-aquila-es` | `aura-2-carina-es` | `aura-2-celeste-es` | `aura-2-diana-es` | `aura-2-estrella-es` | `aura-2-gloria-es` | `aura-2-javier-es` | `aura-2-luciano-es` | `aura-2-nestor-es` | `aura-2-olivia-es` | `aura-2-selena-es` | `aura-2-silvia-es` | `aura-2-sirio-es` | `aura-2-valerio-es` | `aura-2-aurelia-de` | `aura-2-elara-de` | `aura-2-fabian-de` | `aura-2-julius-de` | `aura-2-kara-de` | `aura-2-lara-de` | `aura-2-viktoria-de` | `aura-2-beatrix-nl` | `aura-2-cornelia-nl` | `aura-2-daphne-nl` | `aura-2-hestia-nl` | `aura-2-lars-nl` | `aura-2-leda-nl` | `aura-2-rhea-nl` | `aura-2-roman-nl` | `aura-2-sander-nl` | `aura-2-agathe-fr` | `aura-2-hector-fr` | `aura-2-cesare-it` | `aura-2-cinzia-it` | `aura-2-demetra-it` | `aura-2-dionisio-it` | `aura-2-elio-it` | `aura-2-flavio-it` | `aura-2-livia-it` | `aura-2-maia-it` | `aura-2-melia-it` | `aura-2-ama-ja` | `aura-2-ebisu-ja` | `aura-2-fujin-ja` | `aura-2-izanami-ja` | `aura-2-uzume-ja` (default: `aura-asteria-en`) — AI model used to process submitted text +- `sample_rate` `8000` | `16000` | `24000` | `32000` | `48000` | `8000` | `16000` | `8000` | `16000` | `22050` | `48000` (default: `24000`) — Sample Rate specifies the sample rate for the output audio. Based on the encoding, different sample rates are supported. For some encodings, the sample rate is not configurable +- `speed` number (default: `1`, range: `0.7` to `1.5`) — Speaking rate multiplier that adjusts the pace of generated speech while preserving natural prosody and voice quality. Not yet supported in all languages. + +#### Request Body + +**application/json** + +- `text` string **(required)** — The text content to be converted to speech + +#### Responses + +**200**: Successful text-to-speech transformation +**400**: Invalid Request + +### POST `/v2/speak` + +Flux Text to Speech (batch) + +Synthesize a complete block of text into a single audio response using Deepgram's Flux TTS batch (REST) API. Use this for pre-rendering fixed audio (IVR prompts, notifications, narration) where the whole text is known up front and you don't need incremental playback or interruption. + +#### Query Parameters + +- `callback` string — URL to which we'll make the callback request +- `callback_method` `POST` | `PUT` (default: `POST`) — HTTP method by which the callback request will be made +- `mip_opt_out` boolean (default: `false`) — Opts out requests from the Deepgram Model Improvement Program. Refer to our Docs for pricing impacts before setting this to true. https://dpgr.am/deepgram-mip +- `tag` string | string[] — Label your requests for the purpose of identification during usage reporting +- `bit_rate` `8000` | `16000` | `24000` | `32000` | `40000` | `48000` | integer | integer (default: `48000`) — The bitrate of the audio in bits per second. Choose from predefined ranges or specific values based on the encoding type. +- `container` `none` | `wav` | `wav` | `wav` | `ogg` (default: `wav`) — Container specifies the file format wrapper for the output audio. The available options depend on the encoding type. +- `encoding` `linear16` | `flac` | `mulaw` | `alaw` | `mp3` | `opus` | `aac` (default: `mp3`) — Encoding allows you to specify the expected encoding of your audio output +- `expressivity` `-2` | `-1` | `0` | `1` | `2` (default: `0`) — Expressive range of the generated speech, on a calm-to-animated axis. Accepted values: `-2`, `-1`, `0`, `1`, `2`. `0` (the default) is the voice's tuned delivery and the production-validated setting, with `-2` the calm end of the range and `2` the animated end. Supported on all Flux voices; applies to the whole request. Beta: behavior may change in future model versions, and non-default values increase the risk of hallucinations and pronunciation errors; audition before shipping. An invalid value is rejected with a `400` — `EXPRESSIVITY_OUT_OF_RANGE` for a value outside the range, `EXPRESSIVITY_INCREMENT_INVALID` for a fractional value. See [Expressivity](/docs/tts-expressivity). +- `model` string **(required)** — Flux TTS model used to synthesize the submitted text, in the form `flux-{voice}-{language}` (for example, `flux-alexis-en`). Required; unlike the v1 (Aura) endpoint there is no default and only flux models are accepted. English-only at launch. +- `sample_rate` `8000` | `16000` | `24000` | `32000` | `44100` | `48000` | `8000` | `16000` | `8000` | `16000` | `8000` | `16000` | `22050` | `32000` | `48000` (default: `24000`) — Sample Rate specifies the sample rate for the output audio. Based on the encoding, different sample rates are supported. For some encodings, the sample rate is not configurable +- `speed` number (default: `1`) — Speaking rate multiplier that adjusts the pace of generated speech while preserving natural prosody and voice quality. Accepted values run `0.5` to `1.5` in `0.05` increments. Not yet supported in all languages. +- `priority` `low` — Processing priority for asynchronous (callback) requests. The only supported value is low. + +#### Request Body + +**application/json** + +- `text` string **(required)** — The text content to be converted to speech. The server normalizes and preprocesses the text before synthesis. Inline pause and pronunciation controls are not yet applied; they are stripped from the text before synthesis. + +#### Responses + +**200**: Returns the synthesized audio in the requested encoding as a binary stream. When a `callback` URL is supplied, the request is processed asynchronously and the response body is instead a JSON acknowledgement (Content-Type `application/json`) of the form {"request_id": "..."}, with the audio delivered to the callback URL. Because this endpoint is typed as a binary audio stream, SDK callers that set `callback` receive this JSON acknowledgement through the audio byte iterator as raw bytes and must join the chunks and parse `request_id` themselves. +**400**: Invalid Request. Inline pause and pronunciation controls are not applied and are stripped rather than rejected. + +## WebSocket API + +### WebSocket `/v1/speak` +> Server: `wss://api.deepgram.com` + +Convert text into natural-sounding speech using Deepgram's TTS WebSocket + +#### Connection Parameters + +- `encoding` `linear16` | `mulaw` | `alaw` (default: `linear16`) — Encoding allows you to specify the expected encoding of your audio output for streaming TTS. Only streaming-compatible encodings are supported. +- `mip_opt_out` boolean (default: `false`) — Opts out requests from the Deepgram Model Improvement Program. Refer to our Docs for pricing impacts before setting this to true. https://dpgr.am/deepgram-mip +- `model` `aura-angus-en` | `aura-arcas-en` | `aura-asteria-en` | `aura-athena-en` | `aura-helios-en` | `aura-hera-en` | `aura-luna-en` | `aura-orion-en` | `aura-orpheus-en` | `aura-perseus-en` | `aura-stella-en` | `aura-zeus-en` | `aura-2-amalthea-en` | `aura-2-andromeda-en` | `aura-2-apollo-en` | `aura-2-arcas-en` | `aura-2-aries-en` | `aura-2-asteria-en` | `aura-2-athena-en` | `aura-2-atlas-en` | `aura-2-aurora-en` | `aura-2-callista-en` | `aura-2-cora-en` | `aura-2-cordelia-en` | `aura-2-delia-en` | `aura-2-draco-en` | `aura-2-electra-en` | `aura-2-harmonia-en` | `aura-2-helena-en` | `aura-2-hera-en` | `aura-2-hermes-en` | `aura-2-hyperion-en` | `aura-2-iris-en` | `aura-2-janus-en` | `aura-2-juno-en` | `aura-2-jupiter-en` | `aura-2-luna-en` | `aura-2-mars-en` | `aura-2-minerva-en` | `aura-2-neptune-en` | `aura-2-odysseus-en` | `aura-2-ophelia-en` | `aura-2-orion-en` | `aura-2-orpheus-en` | `aura-2-pandora-en` | `aura-2-phoebe-en` | `aura-2-pluto-en` | `aura-2-saturn-en` | `aura-2-selene-en` | `aura-2-thalia-en` | `aura-2-theia-en` | `aura-2-vesta-en` | `aura-2-zeus-en` | `aura-2-agustina-es` | `aura-2-alvaro-es` | `aura-2-antonia-es` | `aura-2-aquila-es` | `aura-2-carina-es` | `aura-2-celeste-es` | `aura-2-diana-es` | `aura-2-estrella-es` | `aura-2-gloria-es` | `aura-2-javier-es` | `aura-2-luciano-es` | `aura-2-nestor-es` | `aura-2-olivia-es` | `aura-2-selena-es` | `aura-2-silvia-es` | `aura-2-sirio-es` | `aura-2-valerio-es` | `aura-2-aurelia-de` | `aura-2-elara-de` | `aura-2-fabian-de` | `aura-2-julius-de` | `aura-2-kara-de` | `aura-2-lara-de` | `aura-2-viktoria-de` | `aura-2-beatrix-nl` | `aura-2-cornelia-nl` | `aura-2-daphne-nl` | `aura-2-hestia-nl` | `aura-2-lars-nl` | `aura-2-leda-nl` | `aura-2-rhea-nl` | `aura-2-roman-nl` | `aura-2-sander-nl` | `aura-2-agathe-fr` | `aura-2-hector-fr` | `aura-2-cesare-it` | `aura-2-cinzia-it` | `aura-2-demetra-it` | `aura-2-dionisio-it` | `aura-2-elio-it` | `aura-2-flavio-it` | `aura-2-livia-it` | `aura-2-maia-it` | `aura-2-melia-it` | `aura-2-ama-ja` | `aura-2-ebisu-ja` | `aura-2-fujin-ja` | `aura-2-izanami-ja` | `aura-2-uzume-ja` (default: `aura-asteria-en`) — AI model used to process submitted text +- `sample_rate` `8000` | `16000` | `24000` | `32000` | `48000` (default: `24000`) — Sample Rate specifies the sample rate for the output audio. Based on encoding 8000 or 24000 are possible defaults. For some encodings sample rate is not configurable. +- `speed` number (default: `1`, range: `0.7` to `1.5`) — Speaking rate multiplier that adjusts the pace of generated speech while preserving natural prosody and voice quality. Not yet supported in all languages. + +#### Client → Server Messages + +**SpeakV1Text** — Text to convert to audio + + - `type` `Speak` **(required)** — Message type identifier + - `text` string **(required)** — The input text to be converted to speech + +**SpeakV1Flush** — Flush the buffer and receive the final audio for text sent so far + + - `type` `Flush` | `Clear` | `Close` **(required)** — Message type identifier + +**SpeakV1Clear** — Clear the buffer and start a new audio generation. Potentially destructive operation for any text in the buffer + + - `type` `Flush` | `Clear` | `Close` **(required)** — Message type identifier + +**SpeakV1Close** — Flush the buffer and close the connection gracefully after all audio is generated + + - `type` `Flush` | `Clear` | `Close` **(required)** — Message type identifier + +#### Server → Client Messages + +**SpeakV1Audio** — Receive audio chunks as they are generated + +**SpeakV1Metadata** — Receive metadata about the audio generation + + - `type` `Metadata` **(required)** — Message type identifier + - `request_id` string **(required)** — Unique identifier for the request + - `model_name` string **(required)** — Name of the model being used + - `model_version` string **(required)** — Version of the primary model being used + - `model_uuid` string **(required)** — Unique identifier for the primary model used + - `additional_model_uuids` string[] — List of unique identifiers for any additional models used to serve the request + +**SpeakV1Flushed** — Receive metadata about the audio generation + + - `type` `Flushed` | `Cleared` **(required)** — Message type identifier + - `sequence_id` integer **(required)** — The sequence ID of the response + +**SpeakV1Cleared** — Receive metadata about the audio generation + + - `type` `Flushed` | `Cleared` **(required)** — Message type identifier + - `sequence_id` integer **(required)** — The sequence ID of the response + +**SpeakV1Warning** — Receive a warning about the audio generation + + - `type` `Warning` **(required)** — Message type identifier + - `description` string **(required)** — A description of what went wrong + - `code` string **(required)** — Error code identifying the type of error + +### WebSocket `/v2/speak` +> Server: `wss://api.deepgram.com` + +Streaming, turn-based text-to-speech (Flux TTS) built for voice-agent +pipelines. Stream LLM tokens in, speak them to the user, and report +per-turn billing and timing. + + +#### Connection Parameters + +- `model` string — The Flux TTS model used to synthesize speech. Required on every connection. Model strings follow the format `flux-{voice}-{language}` (e.g. `flux-alexis-en`). An Aura model string is rejected on `/v2/speak`; use `/v1/speak` for Aura voices. +- `encoding` `linear16` | `mulaw` | `alaw` (default: `linear16`) — Encoding of the raw output audio. The streaming WebSocket emits raw (non-containerized) audio, so only streaming-compatible encodings are supported. Compressed and containerized encodings (`mp3`, `opus`, `flac`, `aac`) are available on the batch REST transport only. +- `sample_rate` `8000` | `16000` | `24000` | `32000` | `44100` | `48000` — Output sample rate in Hz. With `linear16`, valid values are `8000`, `16000`, `24000`, `32000`, `44100`, and `48000`. With `mulaw` or `alaw`, valid values are `8000` and `16000`. Defaults to the model's native sample rate. +- `speed` number (default: `1`) — Speech-rate multiplier. `1.0` is the model's nominal rate; lower is slower. Accepted values run `0.5` to `1.5` in `0.05` increments. A value outside that range is rejected with `SPEED_OUT_OF_RANGE`; a value inside it but off the `0.05` increment with `SPEED_INCREMENT_INVALID`. Models and languages without runtime speed control reject any value with `SPEED_NOT_SUPPORTED`. +- `expressivity` `-2` | `-1` | `0` | `1` | `2` (default: `0`) — Expressive range of the generated speech, on a calm-to-animated axis. Accepted values: `-2`, `-1`, `0`, `1`, `2`. `0` (the default) is the voice's tuned delivery and the production-validated setting, with `-2` the calm end of the range and `2` the animated end. Supported on all Flux voices. Fixed for the connection — not settable via `Configure`. Beta: behavior may change in future model versions, and non-default values increase the risk of hallucinations and pronunciation errors; audition before shipping. An invalid value fails the connection with a `400` — `EXPRESSIVITY_OUT_OF_RANGE` for a value outside the range, `EXPRESSIVITY_INCREMENT_INVALID` for a fractional value. See [Expressivity](/docs/tts-expressivity). +- `mip_opt_out` boolean (default: `false`) — Opts out requests from the Deepgram Model Improvement Program. Refer to our Docs for pricing impacts before setting this to true. https://dpgr.am/deepgram-mip +- `tag` string | string[] — Label your requests for the purpose of identification during usage reporting + +#### Client → Server Messages + +**SpeakV2Speak** — Send text to be synthesized into the active turn + + - `type` `Speak` **(required)** — Message type identifier + - `text` string **(required)** — The input text to synthesize. Inline pause and pronunciation controls are not yet applied; they are stripped from the text before synthesis. + +**SpeakV2Flush** — End the active turn and generate the remaining audio + + - `type` `Flush` **(required)** — Message type identifier + +**SpeakV2Interrupt** — Cancel the active turn because the user barged in + + - `type` `Interrupt` **(required)** — Message type identifier + - `playback_offset` { type: `time_ms`, value: integer } — How much audio the client had played when the user barged in. Optional: without it the server cannot split the turn's text, so `SpeechInterrupted` omits `text_spoken` and `text_remaining`. + + The offset is cumulative from the start of the *session*, not from the start of the current turn. Each `Interrupt` must advance past the position the previous one established. + +**SpeakV2Configure** — Update synthesis configuration mid-session + + - `type` `Configure` **(required)** — Message type identifier + - `speed` number (default: `1`) — Speech-rate multiplier. `1.0` is the model's nominal rate; lower is slower. Accepted values run `0.5` to `1.5` in `0.05` increments. A value outside that range is rejected with `SPEED_OUT_OF_RANGE`; a value inside it but off the `0.05` increment with `SPEED_INCREMENT_INVALID`. Models and languages without runtime speed control reject any value with `SPEED_NOT_SUPPORTED`. + +**SpeakV2Close** — Gracefully close the connection, draining all remaining and queued audio + + - `type` `Close` **(required)** — Message type identifier + +#### Server → Client Messages + +**SpeakV2Audio** — Receive audio chunks as they are generated + +**SpeakV2Connected** — Receive a connected message on a successful connection + + - `type` `Connected` **(required)** — Message type identifier + - `request_id` string **(required)** — The unique identifier of the `/v2/speak` request + - `model_name` string **(required)** — Resolved model name + - `model_version` string **(required)** — Resolved model version + - `model_uuids` string[] **(required)** — Resolved model UUIDs. A list, because a resolved model may be backed by more than one underlying model. + +**SpeakV2SpeechStarted** — Receive a message marking the start of a new turn, carrying the turn's unique identifier + + - `type` `SpeechStarted` **(required)** — Message type identifier + - `speech_id` string **(required)** — Server-minted identifier for this turn, of the form `dg_sp_<12 hex digits>`. Informational. + +**SpeakV2SpeechMetadata** — Receive per-turn billing and timing after a manual Flush + + - `type` `SpeechMetadata` **(required)** — Message type identifier + - `speech_id` string **(required)** — Server-assigned turn identifier + - `audio_duration_ms` integer **(required)** — Total audio duration produced for this turn, in milliseconds + - `input_character_count` integer **(required)** — Raw input character count for this turn, before text normalization + - `billable_character_count` integer **(required)** — Billable character count for this turn — the input character count with stripped control characters removed. Always less than or equal to `input_character_count`. + - `controls_applied` { pronunciations_applied: integer, breaks_applied: integer, pronunciation_warnings: integer } **(required)** — Counts of the inline controls the server acted on during the turn. Inline pause and pronunciation controls are not applied at launch — support is coming soon — so every count is currently `0`. + +**SpeakV2SpeechInterrupted** — Receive what the user heard, and the interrupted turn's billing, after an Interrupt + + - `type` `SpeechInterrupted` **(required)** — Message type identifier + - `audio_played_ms` integer **(required)** — How much audio the client had played when the interrupt landed, in milliseconds from the start of the session. Echoes the `Interrupt`'s `playback_offset` when one was supplied. Otherwise it is the server's own total, representing the audio that has been generated so far. A client that sends its first `Interrupt` without an offset can use this value as the baseline the next one must advance past. + - `text_spoken` string — The portion of the turn's text the user heard. Omitted when the `Interrupt` carried no `playback_offset`. + - `text_remaining` string — The portion of the turn's text the user did not hear. Omitted when the `Interrupt` carried no `playback_offset`. + - `metadata` { speech_id: string, audio_duration_ms: integer, input_character_count: integer, billable_character_count: integer, controls_applied: { pronunciations_applied: integer, breaks_applied: integer, pronunciation_warnings: integer } } **(required)** — Billing and timing for a single turn. + +**SpeakV2Flushed** — Receive an echo confirming receipt of a manual Flush + + - `type` `Flushed` **(required)** — Message type identifier + - `speech_id` string **(required)** — Server-assigned turn identifier + +**SpeakV2SessionMetadata** — Receive cumulative session totals before the socket closes + + - `type` `SessionMetadata` **(required)** — Message type identifier + - `total_audio_duration_ms` integer **(required)** — Cumulative audio duration produced across the session, in milliseconds. An `Interrupt` rebases this onto the audio the client actually played. + - `total_input_character_count` integer **(required)** — Cumulative raw input character count across the session + - `total_billable_character_count` integer **(required)** — Cumulative billable character count across the session + +**SpeakV2ConfigureSuccess** — Receive confirmation that a Configure was accepted and applied, echoing the applied configuration + + - `type` `ConfigureSuccess` **(required)** — Message type identifier + - `applied` { speed: number } **(required)** — Synthesis configuration. A field is present only when it has been set on this session. + +**SpeakV2ConfigureFailure** — Receive notice that a Configure was rejected or failed to apply; the prior configuration is retained + + - `type` `ConfigureFailure` **(required)** — Message type identifier + - `code` `SPEED_OUT_OF_RANGE` | `SPEED_INCREMENT_INVALID` | `SPEED_NOT_SUPPORTED` | `INTERNAL_ERROR` **(required)** — Failure code, in `SCREAMING_SNAKE_CASE`. `SPEED_OUT_OF_RANGE`: outside the range the model publishes. `SPEED_INCREMENT_INVALID`: inside the published range but off the `0.05` increment. `SPEED_NOT_SUPPORTED`: this model or language has no runtime speed control at all. `INTERNAL_ERROR`: the configuration was acceptable but the server could not apply it — unlike the others, a server-side failure rather than a statement about the request. + - `field` `speed` — The configuration field the failure is about. Absent when the failure is not tied to one field. + - `value` number — The rejected value for `field`. Absent when there is no offending value to echo — `SPEED_NOT_SUPPORTED` names the field but carries no value, because the rejection is a property of the model. + - `description` string **(required)** — A human-readable description of the failure + +**SpeakV2Warning** — Receive a warning; synthesis continues and the connection is unaffected + + - `type` `Warning` **(required)** — Message type identifier + - `code` string **(required)** — Warning code identifying the condition, in `SCREAMING_SNAKE_CASE`. + + Turn-scoped codes: `NO_ACTIVE_SPEECH` (a speech-scoped message arrived with no active turn), `NO_SYNTHESIZABLE_TEXT` (the turn's text was entirely whitespace or punctuation, so it produced no audio and is completed with a zero-duration `SpeechMetadata`), and `SYNTHESIS_RETRYING` (a synthesis request failed and is being retried). + + Inline-control codes are reserved and not currently emitted, because inline pause and pronunciation controls are not yet applied: `BREAKS_LIMIT_EXCEEDED` (too many pause controls, or two pauses with no intervening text), `BREAK_TOKENS_OUT_OF_RANGE` (pause durations outside the range the model supports), `BREAK_TOKENS_WITH_INVALID_INCREMENTS` (pause durations off the model's supported increment), `PRONUNCIATION_WARNINGS` (a pronunciation override contained invalid IPA), `PRONUNCIATION_TOO_LONG` (an IPA string exceeded the length limit), `PRONUNCIATIONS_LIMIT_EXCEEDED` (too many pronunciation controls in one turn). + + Interrupt-scoped codes, each meaning the `Interrupt` was ignored: `NO_AUDIO_GENERATED` (the session has produced no audio yet, so there is nothing to interrupt), `INTERRUPT_IN_PROGRESS` (an earlier `Interrupt` is still being processed — at most one is handled at a time), `INVALID_INTERRUPT_OFFSET` (the `playback_offset` did not advance past the position a prior interrupt established). + - `description` string **(required)** — A human-readable description of the warning + +**SpeakV2Error** — Receive a fatal error message followed by a WebSocket close + + - `type` `Error` **(required)** — Message type identifier + - `code` `MESSAGE-0000` | `DATA-0000` | `DATA-0002` | `BIG-0000` | `NET-0000` | `NET-0001` | `NET-0002` | `NET-0003` | `NET-0004` **(required)** — A code identifying the error, e.g. `MESSAGE-0000` or `NET-0000`. + - `description` string **(required)** — Prose description of the error diff --git a/.agents/skills/audio-intelligence/SKILL.md b/.agents/skills/audio-intelligence/SKILL.md new file mode 100644 index 0000000..d735bc9 --- /dev/null +++ b/.agents/skills/audio-intelligence/SKILL.md @@ -0,0 +1,144 @@ +--- +name: audio-intelligence +description: > + Analyze what was said in audio, not just transcribe it. Use when a task mentions + "audio intelligence", "sentiment", "sentiment analysis", "summarize a recording", + "summarization", "topics", "topic detection", "intents", "intent recognition", + "entity detection", "detect entities", "extract names and amounts from a call", + or "analyze a call recording". These are five query parameters layered on the + speech-to-text endpoint /v1/listen (summarize, sentiment, topics, intents, + detect_entities), not a separate API. Covers the prerecorded-only and English-only + limits, where each result lives in the JSON, and the errors you get when you cross + a limit. Routes to the api, docs, recipes, starters, and per-language SDK skills. +--- + +# Deepgram Audio Intelligence + +Audio intelligence is not a separate endpoint. It is five query parameters on `/v1/listen`, the +same endpoint that returns the transcript. You get the transcript and the analysis in one response, +from one API call. This skill gets a verified request working and states the limits precisely. + +## Decide first + +- **Your input is audio** (a file, a URL, a call recording) and you want analysis: stay here, use + `POST https://api.deepgram.com/v1/listen`. +- **Your input is already text** (a transcript, an email, a chat log, a support ticket): the + parameters below do not apply. Use the Read API, `POST /v1/read`. Open the `text-intelligence` skill. +- **You only want the transcript**: drop these parameters and open the `speech-to-text` skill. +- **You are streaming live audio**: only `detect_entities` is available. See the matrix. + +## Feature matrix + +| Parameter | Prerecorded | Streaming (`wss`) | Language | +|---|---|---|---| +| `summarize=v2` (or `summarize=true`) | yes | **no** | English only, enforced with a 400 | +| `sentiment=true` | yes | **no** | English only | +| `topics=true` | yes | **no** | English only | +| `intents=true` | yes | **no** | English only | +| `detect_entities=true` | yes | **yes** | English only | + +`detect_entities` is the odd one out twice over: it is the only feature that works on the live +socket, and it is the only one that does **not** exist on the Read API. Streaming entity detection +runs on Nova, Nova-2, Nova-3, and Enhanced; it is not available on Base models or on Flux. [2] + +## Verified request + +```bash +curl -s -X POST 'https://api.deepgram.com/v1/listen?model=nova-3&smart_format=true&summarize=v2&sentiment=true&topics=true&intents=true&detect_entities=true' \ + -H "Authorization: Token $DEEPGRAM_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{"url":"https://dpgr.am/spacewalk.wav"}' +``` + +Returns 200. The analysis is scattered across the response, not collected in one place: + +| Result | Path | +|---|---| +| Summary | `results.summary.short` (with `results.summary.result` = `"success"`) | +| Sentiment per segment | `results.sentiments.segments[]` — `text`, `start_word`, `end_word`, `sentiment`, `sentiment_score` | +| Sentiment overall | `results.sentiments.average` — `sentiment`, `sentiment_score` | +| Topics | `results.topics.segments[].topics[]` — `topic`, `confidence_score` | +| Intents | `results.intents.segments[].intents[]` — `intent`, `confidence_score` | +| Entities | `results.channels[0].alternatives[0].entities[]` — `label`, `value`, `confidence`, `start_word`, `end_word` | + +Note the plural: the parameter is `sentiment`, the result key is `sentiments`. `metadata` gains a +`summary_info`, `sentiment_info`, `topics_info`, and `intents_info` block per enabled feature, each +with `model_uuid`, `input_tokens`, and `output_tokens`. Entity labels come back uppercased +(`NAME`, `ORGANIZATION`, `LOCATION_CITY`, `MONEY`, `DATE_INTERVAL`); Deepgram documents over 50 +types. [6] + +On the live socket, `detect_entities=true` adds a **top-level** `entities` array to each `Results` +message, next to `channel` — not inside `channel.alternatives[0]`. Same field shape as above. + +## Narrowing topics and intents + +`custom_topic` and `custom_intent` (repeatable) add your own labels; `custom_topic_mode` and +`custom_intent_mode` take `extended` (default, your labels plus the model's) or `strict` (your +labels only). `strict` returns `"segments": []` whenever nothing matches your list, which looks +like a broken request but is not. Start with `extended`. + +## Common mistakes + +1. **Expecting these to work on streaming.** Each failure mode is different, which makes this + confusing. `summarize` fails the WebSocket handshake with 400 `"Summarization is not available + for streaming."`. `topics` and `intents` fail it with 403 + `{"err_code":"UNAUTHORIZED_FEATURES_REQUESTED","err_msg":"Project does not have access to the + requested feature/s [\"topics\"]."}` — which reads as a permissions problem but is not one: the + same key's prerecorded `topics` requests succeed, and the docs matrix lists streaming as + unsupported for both. [2] Do not go asking for an entitlement. `sentiment` is worse still: the + handshake **succeeds** and no sentiment is ever returned. Only `detect_entities` works. +2. **Expecting a non-English request to fail loudly.** It does not. With `language=es` on real + Spanish audio, `sentiment`, `topics`, `intents`, and `detect_entities` return **HTTP 200** with + the transcript, the analysis keys silently absent, and the reason only in `metadata.warnings`: + `[{"parameter":"sentiment","type":"unsupported_language","message":"Sentiment is only supported + for English."}]`, plus `"Topics are only supported for English."`, `"Intents are only supported + for English."`, and `"Entity detection is only supported for English."`. Read + `metadata.warnings` before you conclude the model found nothing. +3. **Assuming `summarize` behaves the same way.** It is the exception: non-English is a hard 400, + `{"err_code":"Bad Request","err_msg":"Summarization v2 not supported for non-English languages"}`. + `language=multi` gets the same 400. +4. **`language=multi` as a workaround.** It is not one. `multi` returns the analysis when the + detected speech is English and drops it with the same `metadata.warnings` when it is not, so the + same request succeeds or silently degrades depending on what the caller said. +5. **`summarize=v1`.** Returns 400 `"To use the summarize feature, please use 'summarize=true' or + 'summarize=v2'. The 'summarize=v1' parameter is deprecated."` Use `v2`; `true` is accepted and + returns the same `summary.short` shape. +6. **Looking for `results.summary.text`.** That is the Read API's shape. On `/v1/listen` the + summary is at `results.summary.short`. +7. **Putting any of these on Flux.** `/v2/listen` rejects all five at the handshake with 400 + `{"err_code":"INVALID_QUERY_PARAMETER","err_msg":"Unknown query parameters: detect_entities"}`, + and the same message naming `summarize`, `sentiment`, `topics`, or `intents`. Transcribe with + Flux, then send the transcript to `/v1/read`. +8. **Reaching for these to mask PII.** Detection returns entities, it does not remove them. Use + `redact` for that, which is a speech-to-text parameter. [6] + +## Pricing + +Enabling these features changes what a request costs. Rates and the billing model change, so read + rather than any figure quoted in a skill. + +## Use a different skill when + +- Your input is text rather than audio: `text-intelligence` skill (`/v1/read`). +- You want every parameter and the full response schema: `api` skill, `references/listen.md`. +- You only need transcription, diarization, redaction, or captions: `speech-to-text` skill. +- You want a runnable demo app: `starters` skill. Note there is no `audio-intelligence` starter; the + `text-intelligence` feature (13 languages) is the Read API app. +- You want a snippet under 50 lines: `recipes` skill, "Audio Intelligence `v1`" — `summarize`, + `sentiment`, `topics`, `intents`, `entities`, in Python, JavaScript, Go, .NET, Java, Rust, and the + CLI. [7] +- You want language-idiomatic SDK code: install `deepgram-{js,python,java,go,rust,dotnet}-audio-intelligence` + from the matching SDK repository (`npx skills add deepgram/deepgram-python-sdk`, and so on). +- You want to find a docs page: `docs` skill. You want the docs in your editor: `setup-mcp` skill. + +## Sources + +1. https://developers.deepgram.com/docs/audio-intelligence (getting started) +2. https://developers.deepgram.com/docs/stt-intelligence-feature-overview (the prerecorded/streaming/language matrix, and the streaming entity-detection model footnote) +3. https://developers.deepgram.com/docs/summarization, https://developers.deepgram.com/docs/sentiment-analysis, https://developers.deepgram.com/docs/topic-detection, https://developers.deepgram.com/docs/intent-recognition +4. https://developers.deepgram.com/docs/detect-entities +5. https://developers.deepgram.com/docs/language and https://developers.deepgram.com/docs/models-languages-overview +6. https://developers.deepgram.com/docs/supported-entity-types (over 50 types; `redact` for removal) +7. https://github.com/deepgram/recipes/blob/main/COVERAGE.md ("Audio Intelligence `v1`" section) +8. https://developers.deepgram.com/reference/speech-to-text/listen-pre-recorded and https://developers.deepgram.com/reference/speech-to-text/listen-streaming +9. https://developers.deepgram.com/docs/errors and https://deepgram.com/pricing diff --git a/.agents/skills/browser-agent/SKILL.md b/.agents/skills/browser-agent/SKILL.md new file mode 100644 index 0000000..857a893 --- /dev/null +++ b/.agents/skills/browser-agent/SKILL.md @@ -0,0 +1,198 @@ +--- +name: browser-agent +description: > + Run a Deepgram voice agent in the browser with the four Browser Agent SDK packages: + @deepgram/agents (core WebSocket session, mic, player), @deepgram/react (AgentProvider + and hooks), @deepgram/ui (pre-built React components), and @deepgram/agents-widget + (drop-in, no framework). Use when a task says "browser voice agent", "voice widget", + "embed a voice agent", "react voice agent", "@deepgram/react", "@deepgram/ui", + "agents-widget", "AgentProvider", "useAgentState", "useDeepgramAgent", "AgentSession", + "tokenFactory", "Orb", "voice agent on my website", or asks how to keep a Deepgram API + key out of client-side code. Picks the layer, gets one path running, and routes to the + voice-agent skill for the WebSocket contract underneath. +--- + +# Deepgram Browser Agent SDK + +Deepgram runs listen, think, and speak behind one WebSocket and handles the turn-taking between them. These four packages get a browser onto that socket; each wraps the one below. Pick a layer by how much UI you want to build yourself, not by package name. [1] + +## Pick the layer first + +| Package | When | Install | +|---|---|---| +| `@deepgram/agents-widget` | You want a voice agent on a page today, with no framework and no build step, and you can live with its UI. It ships six layouts (`sidebar`, `floating`, `inline`, `button`, `embedded`, `orb`), themed with CSS custom properties. | `@deepgram/agents-widget` | +| `@deepgram/ui` | React app, and the conversation view, orb, waveform, and mic/speaker buttons listed below are close enough to your design. Retheme them with CSS custom properties. | `@deepgram/ui` only. See Common mistakes 3 | +| `@deepgram/react` | React app, you build every pixel. Provider plus focused hooks manage the connection, mic, playback, transcript, and client tools. | `@deepgram/react @deepgram/agents` | +| `@deepgram/agents` | You are on Vue, Svelte, Angular, vanilla JS, or another non-React runtime. `AgentSession`, `AgentMicrophone`, and `AgentPlayer` give you the pieces; you wire the events. | `@deepgram/agents` | + +Each layer re-exports the layer below, so `@deepgram/ui` alone gives you `AgentProvider`, the `@deepgram/agents` types, and all ten hooks: `useAgentClientTool`, `useAgentContext`, `useAgentControls`, `useAgentConversation`, `useAgentMicrophone`, `useAgentMode`, `useAgentPlayer`, `useAgentSession`, `useAgentState`, and `useDeepgramAgent`. [7] + +## Versions and stability + +All four are pre-1.0, and `latest` is the only dist-tag on each. `@deepgram/ui`'s README puts it plainly: "This library is pre-1.0. Interfaces may change between minor versions, and releases are cut as the library evolves rather than on a fixed schedule." So resolve the versions against the registry before you install, and resolve them again before you trust a version-specific statement anywhere below: + +```bash +for p in @deepgram/agents @deepgram/agents-widget @deepgram/react @deepgram/ui; do npm view "$p" version; done +npm view @deepgram/ui dependencies # and @deepgram/agents-widget, to check the ranges below +``` + +Where the registry disagrees with this skill, the registry wins. The table below is the pinned set the export names and dependency ranges in this skill describe. [3][4][5] + +| Package | Version | Repo | +|---|---|---| +| `@deepgram/agents` | 0.1.2 | `deepgram/agent` (`packages/sdk`) | +| `@deepgram/agents-widget` | 0.1.8 | `deepgram/agent` (`packages/widget`) | +| `@deepgram/react` | 0.2.0 | `deepgram/react` | +| `@deepgram/ui` | 0.1.6 | `deepgram/ui` | + +Declared runtime dependencies, as published. `@deepgram/agents` depends on `@deepgram/sdk` 5.9.0. `@deepgram/react@0.2.0` depends on `@deepgram/agents ^0.1.2` and takes `react` and `react-dom` `>=18.0.0` as peers. `@deepgram/ui@0.1.6` depends on `@deepgram/react ^0.1.0`, `@deepgram/agents ^0.1.1`, and Radix and Tailwind helpers. `@deepgram/agents-widget@0.1.8` depends on `@deepgram/react ^0.1.0`, `@deepgram/ui ^0.1.4`, `@deepgram/agents ^0.1.2`, and `preact`, all bundled into its own build. Those `^0.1.0` ranges exclude `@deepgram/react@0.2.0`, which is what makes mistake 3 below possible. [6] + +## Browser auth: never the API key + +A browser `WebSocket` cannot set request headers, so the SDK sends a short-lived bearer token as the `Sec-WebSocket-Protocol` handshake value. You supply that token through `tokenFactory`, which the SDK calls before every connect and every reconnect, so a few seconds of TTL is enough. The token only has to be valid at the handshake: once the socket is open, the token expiring does not close it, and a 30-second token is fine for an hour-long call. [1][2] + +Mint them on your own server. `POST https://api.deepgram.com/v1/auth/grant` needs an API key with Member or higher authorization and returns `{"access_token":"...","expires_in":30}`. Its tokens work on the voice APIs but not on the Manage APIs: + +```js +// Server. The API key never leaves this process. +app.get("/api/deepgram-token", async (_req, res) => { + const r = await fetch("https://api.deepgram.com/v1/auth/grant", { + method: "POST", + headers: { Authorization: `Token ${process.env.DEEPGRAM_API_KEY}`, + "Content-Type": "application/json" }, + body: JSON.stringify({ ttl_seconds: 30 }), + }); + const { access_token } = await r.json(); + res.set("Cache-Control", "no-store").send(access_token); +}); +``` + +Pass it as `auth: { tokenFactory: () => fetch("/api/deepgram-token").then(r => r.text()) }` in React and the SDK, or as a top-level `tokenFactory` in the widget. Put your own session check in front of that route: anyone who can call it can open an agent session billed to you. The `apiKey` auth mode exists for server-side use and local experiments only; in a browser bundle it is a published credential. [1][2] + +## `@deepgram/agents-widget`: one call + +```js +import { init } from "@deepgram/agents-widget"; + +const teardown = init({ + tokenFactory: () => fetch("/api/deepgram-token").then((r) => r.text()), + agent: "YOUR_AGENT_ID", // a Reusable Agent Configuration UUID, or a full settings object + layout: "floating", + placement: "bottom-right", +}); +// teardown() unmounts and removes injected styles; call it on SPA route change +``` + +`init` is the package's only function export; everything else it ships is type declarations. For a no-build page, load the UMD bundle and call `DeepgramAgent.init(...)`; the global is `DeepgramAgent`. `https://cdn.deepgram.com/widgets/v0.1.8/widget.umd.js` serves the same bundle as npm 0.1.8, 414,517 bytes minified and roughly 91 KB gzipped. The same CDN answers a `latest` path, which moves with each release, so pin the `v`-prefixed version in production. [2][8] + +## `@deepgram/react`: provider and hooks + +```tsx +import { AgentProvider, useAgentState, useAgentConversation } from "@deepgram/react"; + +const config = { + auth: { tokenFactory: () => fetch("/api/deepgram-token").then((r) => r.text()) }, + agent: { think: { provider: { type: "open_ai" as const, model: "gpt-4o-mini" } } }, +}; + +export default function App() { + return ; +} + +function Agent() { + const { state, start, stop } = useAgentState(); // idle | connecting | connected | reconnecting | disconnected + const { conversation } = useAgentConversation(); + const onStart = async () => { try { await start(); } catch (e) { console.error(e); } }; + return ( + <> + + {conversation.map((e) =>

{e.role}: {e.content}

)} + + ); +} +``` + +The other hooks: `useAgentMode` (`idle`/`listening`/`thinking`/`speaking`), `useAgentMicrophone`, `useAgentPlayer`, `useAgentControls` (grouped lifecycle, messaging, runtime settings, mute), `useAgentClientTool` (register a function-call handler scoped to the component), `useAgentContext`, `useAgentSession` (the raw `AgentSession`), and `useDeepgramAgent` (no provider needed). `config`, `playerSampleRate`, and the initial `autoStart` are read once and pinned for the provider's lifetime. Change a connected agent with the runtime controls, not by mutating `config`. [4] + +## `@deepgram/ui`: components + +```tsx +import { AgentProvider, AgentConversation, AgentTextInput, AgentStartButton, Orb } from "@deepgram/ui"; +import "@deepgram/ui/styles.css"; + + +
+ + + + +
+
+``` + +Components: `AgentStatus`, `AgentConversation`, `AgentMessage`, `AgentTextInput`, `AgentMicrophoneButton`, `AgentSpeakerButton`, `AgentStartButton`, plus `VoiceButton`, `Orb`, `LiveWaveform`, `BarVisualizer`, `MicSelector`, and `Response`. Styling is Tailwind v4 compiled into `@deepgram/ui/styles.css` and scoped to `[data-dg-agent]`; tokens are shadcn `--color-*` names you override on any `[data-dg-agent]` ancestor. `data-dg-scheme="dark"` on the same element forces dark; without it, components follow `prefers-color-scheme`. Install from npm rather than through a shadcn registry: `deepgram/ui` builds one, but `@deepgram/ui-registry` is a private package and `ui.deepgram.com` serves no registry JSON, so `npx shadcn add` against it returns 404. [5] + +## `@deepgram/agents`: any framework + +```js +import { AgentSession, AgentMicrophone, AgentPlayer } from "@deepgram/agents"; + +const session = new AgentSession({ + auth: { tokenFactory: () => fetch("/api/deepgram-token").then((r) => r.text()) }, + agent: "YOUR_AGENT_ID", +}); +const player = new AgentPlayer(); // default output 24000 Hz +const mic = new AgentMicrophone((data) => session.sendAudio(data)); // default capture 16000 Hz + +session.on("audio", (chunk) => player.queue(chunk)); +session.on("user-started-speaking", () => player.interrupt()); // barge-in +session.on("conversation-text", (m) => console.log(`${m.role}: ${m.content}`)); + +await session.connect(); +await mic.start(); +``` + +`AgentSession` handles the `Welcome`/`Settings`/`SettingsApplied` handshake, buffers mic frames until `SettingsApplied`, sends `KeepAlive`, and reconnects with jittered exponential backoff (`reconnect.maxAttempts` default 8). Runtime methods mirror the protocol: `updateListen`, `updateSpeak`, `updateThink`, `updatePrompt`, `injectUserMessage`, `injectAgentMessage`, `sendFunctionCallResponse`. Events are the protocol messages in kebab-case plus `audio`, `connecting`, `connected`, `reconnecting`, `disconnected`, `sdk-error`. `AgentMicrophone` and `AgentPlayer` expose `getInputVolume` and `getOutputVolume`, plus `getInputByteFrequencyData` and `getOutputByteFrequencyData`, for visualizers. [3] + +## Upgrade `@deepgram/react` 0.1 to 0.2 + +0.2.0 shipped 2026-09-10 and is the only breaking release so far. Its five changes: [9] + +1. `AgentMode` now includes `"thinking"`. Fix exhaustive `switch`es and mode-to-label maps. +2. `AgentContextValue` and the hook result types now require new members. Consuming components need no change; typed mocks, wrappers, and hand-written implementations of those interfaces do. +3. `registerClientTool()` now returns an unsubscribe function. Ignoring it still works; store and call it when registering outside a component lifecycle. `useAgentClientTool()` still unregisters on unmount. +4. `useDeepgramAgent().start()` now clears `conversation` before connecting. Keep any transcript that must outlive a restart outside the hook. Manual `start()` now rejects on failure, so handle the promise; automatic and reconnect failures go to `onSdkError`. +5. `useAgentControls()`, `useAgentConversation()`, and `useDeepgramAgent()` now expose runtime controls, among them `updatePrompt` and `sendAgentMessage`, that update a connected agent without recreating the provider. `onListenUpdated`, `onPromptUpdated`, `onSpeakUpdated`, and `onThinkUpdated` fire when the server acknowledges the change. + +## Common mistakes + +1. Shipping the API key to the browser. `{ auth: { apiKey } }` in client-side code publishes a credential anyone can bill against. Use `tokenFactory` against a route you gate. [1] +2. Setting the token lifetime with `ttl` in the `/v1/auth/grant` body. The field is `ttl_seconds`. A body of `{"ttl":300}` is accepted and ignored, and the response comes back `expires_in: 30`; `{"ttl_seconds":300}` returns `expires_in: 300`. The endpoint ignores any field it does not recognize and still answers HTTP 200, so read `expires_in` in the response rather than trusting the field name you sent. [2] +3. Installing `@deepgram/react@0.2.0` next to `@deepgram/ui@0.1.6` and importing the provider from one and the hooks from the other. `@deepgram/ui` declares `@deepgram/react ^0.1.0`, so npm nests a second copy at 0.1.0 and the two packages build separate React contexts: `AgentProvider` from `@deepgram/ui` is not the same function as `AgentProvider` from `@deepgram/react`, and a hook that reads the other context throws "used outside AgentProvider". Either import everything from `@deepgram/ui` alone, or pin one copy with `"overrides": { "@deepgram/react": "0.2.0" }` in npm, `overrides` in pnpm, or `resolutions` in yarn, which dedupes the tree and makes both imports resolve to one module. [6] +4. Forgetting `import "@deepgram/ui/styles.css"` or the `data-dg-agent` attribute on a wrapper. Every `@deepgram/ui` token is scoped to `[data-dg-agent]`, so without it the components render unstyled. [5] +5. Never calling `player.interrupt()` on `user-started-speaking` in a raw-SDK build. Deepgram stops generating, but your queued audio keeps talking over the caller. Only the raw SDK leaves this to you: `@deepgram/react`'s provider already interrupts the player on that event, and the UI and widget layers inherit it. [3][4] +6. Mismatching sample rates. `AgentPlayer`'s `sampleRate` (default 24000) must equal `audio.output.sample_rate` in your agent settings, and `AgentMicrophone`'s (default 16000) must equal `audio.input.sample_rate`. [3] +7. Leaving `latest` in a production CDN `