diff --git a/.gitattributes b/.gitattributes index 8f64a622..a86c5d21 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,2 +1,3 @@ # output.py is CRLF on main and ruff format keeps it that way; let git diff --check accept it. packages/deepctl-core/src/deepctl_core/output.py whitespace=cr-at-eol +packages/deepctl-core/tests/unit/fixtures/legacy_v03/** -text diff --git a/README.md b/README.md index 87dd317b..e9052ec6 100644 --- a/README.md +++ b/README.md @@ -319,7 +319,13 @@ These commands abort if the filesystem cannot provide locking or no-replace dire If you edit a deepctl folder, or add a file or link to it, deepctl leaves it alone: `update` stops without changing anything and `remove` won't delete it. To get updates again, rename or move your edited copy, or delete it yourself. Opening a skill folder in Finder or Explorer can add `.DS_Store`, `Thumbs.db` or `desktop.ini`, which counts as an edit. -Files from deepctl 0.3.x, such as `~/.claude/commands/deepgram/*.md` and the rules files like `~/.cursor/rules/deepctl.mdc`, are currently kept, and `dg skills remove` doesn't delete them. +#### Upgrading from deepctl 0.2.16 through 0.3.x + +On macOS and Linux, once `dg skills install` or `update` (or `dg login`, or `dg plugin install`, `update` or `remove`) has installed a tool's skill folders, deepctl removes the files older deepctl wrote for that tool, and prints which ones on stderr: `~/.claude/commands/deepgram/*.md`, `~/.cursor/rules/deepctl.mdc`, `~/.cline/rules/deepctl.md`, and the section between the `` lines in `~/.codex/instructions.md`, `~/.gemini/GEMINI.md` and `~/.opencode/agents.md`. In those three files, everything between the two marker lines is removed, including anything you changed there; the rest of the file is kept, apart from the blank line deepctl added before its section. Deepctl keeps the original file next to it as `.deepctl-kept-v03-*`; compare it for any late write from an open editor, then delete it if you do not need it. + +deepctl removes a file only if its content is exactly a deepgram/skills version that older deepctl copied, even if you made that copy yourself. It keeps a file you edited, a link, a file in a folder reached through a link (such as a dotfiles `~/.claude`), and a section that is incomplete, repeated or in a read-only or locked file, and warns once about each file 0.3.x recorded. An I/O failure, such as permission denied, is retried on the next install or update, with a warning for files 0.3.x recorded. Amazon Q Developer and Aider files (`~/.amazonq/rules/deepctl.md`, `~/.deepctl/skills/deepctl-conventions.md`) are kept, because those tools have no skill folders; if you delete Aider's, also remove its entry under `read:` in `~/.aider.conf.yml`. + +On Windows, deepctl leaves this 0.3.x content untouched because a parent folder could become a junction during cleanup; remove it yourself if you do not need it. While deepctl works on a file on macOS or Linux, it moves it to `.deepctl-v03-` in the same folder. If deepctl is interrupted, the next install or update puts the file back and says so. If both the file and its `.deepctl-v03-` are there, as after you saved the file while deepctl was editing it, deepctl changes neither and names both on each run until you compare them and delete the `.deepctl-v03-` copy; it does the same if `.deepctl-v03-` is a link or folder deepctl didn't make. A `.deepctl-v03-*.tmp` file is an unused draft; delete it. ### Starter Apps diff --git a/packages/deepctl-cmd-login/tests/unit/test_login_legacy_v03.py b/packages/deepctl-cmd-login/tests/unit/test_login_legacy_v03.py new file mode 100644 index 00000000..10a98da5 --- /dev/null +++ b/packages/deepctl-cmd-login/tests/unit/test_login_legacy_v03.py @@ -0,0 +1,75 @@ +"""dg login's skills step runs the deepctl 0.3.x cleanup and prints it on stderr.""" + +import os +import shutil +import sys +from pathlib import Path +from unittest.mock import patch + +import click +import pytest +from deepctl_cmd_login import command as login_module +from deepctl_cmd_login.command import LoginCommand +from deepctl_core import output, skill_bundle +from deepctl_core import skill_generator as sg +from deepctl_core.skill_bundle import RepoSkill + +FIX = Path(__file__).parents[3] / "deepctl-core" / "tests" / "unit" / "fixtures" +FIX = FIX / "legacy_v03" +NAMES = ("api", "docs", "setup-mcp", "starters") +pytestmark = pytest.mark.skipif( + os.name == "nt", reason="legacy cleanup is intentionally disabled on Windows" +) + + +@pytest.fixture(autouse=True) +def home(tmp_path, monkeypatch): + home = tmp_path / "home" + home.mkdir() + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.setenv("HOME", str(home)) + monkeypatch.setenv("USERPROFILE", str(home)) + monkeypatch.setattr(sg, "_SKILLS_DIR", home / ".deepctl" / "skills") + monkeypatch.setattr(sg, "_STATE_FILE", home / ".deepctl" / "skills" / "skills.json") + for con in (output.console, output.stderr_console, login_module.console): + monkeypatch.setattr(con, "_width", 400) + monkeypatch.delenv(skill_bundle.REF_ENV_VAR, raising=False) + monkeypatch.setattr(shutil, "which", lambda name: None) + + def fetch(ref=None): + skills = [] + for name in NAMES: + folder = tmp_path / "bundle" / "skills" / name + folder.mkdir(parents=True, exist_ok=True) + (folder / "SKILL.md").write_bytes(f"---\nname: {name}\n---\n".encode()) + skills.append(RepoSkill(name, folder)) + return skills + + monkeypatch.setattr(skill_bundle, "fetch_skill_bundle", fetch) + saved = dict(output._output_config) + yield home + output._output_config.clear() + output._output_config.update(saved) + + +@pytest.mark.parametrize("agentic", [False, True]) +def test_login_skills_step_cleans_up_on_stderr(monkeypatch, capsys, agentic): + output._output_config.update(agentic=agentic, format="default", quiet=False) + files = [] + for n in NAMES: + path = Path.home() / ".claude" / "commands" / "deepgram" / f"{n}.md" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes((FIX / f"{n}.md").read_bytes()) + files.append(path) + # Login offers skills only with no record, so these files are unrecorded + # (skills.json was lost or deleted): the bytes are the proof, not the record. + capsys.readouterr() + monkeypatch.setattr(sys.stdout, "isatty", lambda: True, raising=False) + cmd = LoginCommand() + cmd._guided = True + with patch.object(login_module.Prompt, "ask", return_value="all"): + cmd._maybe_prompt_skills_setup() + out, err = (click.unstyle(s) for s in capsys.readouterr()) + assert not any(p.exists() for p in files) + assert "0.3.x" not in out + assert "Removed deepctl 0.3.x files for Claude Code" in " ".join(err.split()) diff --git a/packages/deepctl-cmd-plugin/tests/unit/test_plugin_command.py b/packages/deepctl-cmd-plugin/tests/unit/test_plugin_command.py index 70e7b9b1..f8e6aaf6 100644 --- a/packages/deepctl-cmd-plugin/tests/unit/test_plugin_command.py +++ b/packages/deepctl-cmd-plugin/tests/unit/test_plugin_command.py @@ -827,7 +827,9 @@ def test_refresh_second_tool_failure_keeps_first_recorded_warns_once_exit_zero( def test_refresh_keeps_03x_and_hint_only_records_byte_identical( self, home, bundle, capsys ): - old = Path.home() / ".claude" / "commands" / "deepgram" / "api.md" + old = Path.home() / ".claude" / "commands" / "deepgram" / "setup-mcp.md" + old.parent.mkdir(parents=True) + old.write_bytes(b"mine") # A folder for setup-mcp never lands: kept. rule = Path.home() / ".amazonq" / "rules" / "deepctl.md" conf = Path.home() / ".aider.conf.yml" legacy = { @@ -837,11 +839,19 @@ def test_refresh_keeps_03x_and_hint_only_records_byte_identical( } write_state({"installed_skills": legacy}) _, err = refresh(capsys) - assert warnings(err) == [] state = sg.get_skills_state() - assert state["installed_skills"] == legacy + if os.name == "nt": + assert len(warnings(err)) == 1 + assert "does not clean 0.3.x content on Windows" in warnings(err)[0] + assert str(old) not in state["installed_skills"]["claude"]["paths"] + assert "v03" not in state["skill_folders"]["claude"] + else: + assert warnings(err) == [] + assert state["installed_skills"] == legacy assert set(state["skill_folders"]) == {"claude"} - assert state["skill_folders"]["claude"]["v03"] is True + assert (state["skill_folders"]["claude"].get("v03") is True) != ( + os.name == "nt" + ) assert (gen("claude").skills_root() / "api" / "SKILL.md").is_file() assert bundle == [REF] diff --git a/packages/deepctl-cmd-plugin/tests/unit/test_plugin_legacy_v03.py b/packages/deepctl-cmd-plugin/tests/unit/test_plugin_legacy_v03.py new file mode 100644 index 00000000..518e38e8 --- /dev/null +++ b/packages/deepctl-cmd-plugin/tests/unit/test_plugin_legacy_v03.py @@ -0,0 +1,94 @@ +"""dg plugin's skills refresh runs the deepctl 0.3.x cleanup on stderr only.""" + +import json +import os +import shutil +from pathlib import Path +from unittest.mock import MagicMock, patch + +import click +import pytest +from click.testing import CliRunner +from deepctl_cmd_plugin import command as plugin_module +from deepctl_cmd_plugin.command import PluginCommand +from deepctl_cmd_plugin.models import PluginOperationResult +from deepctl_core import output, skill_bundle +from deepctl_core import skill_generator as sg +from deepctl_core.skill_bundle import RepoSkill + +FIX = Path(__file__).parents[3] / "deepctl-core" / "tests" / "unit" / "fixtures" +FIX = FIX / "legacy_v03" +NAMES = ("api", "docs", "setup-mcp", "starters") +pytestmark = pytest.mark.skipif( + os.name == "nt", reason="legacy cleanup is intentionally disabled on Windows" +) + + +@pytest.fixture(autouse=True) +def home(tmp_path, monkeypatch): + home = tmp_path / "home" + home.mkdir() + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.setenv("HOME", str(home)) + monkeypatch.setenv("USERPROFILE", str(home)) + monkeypatch.setattr(sg, "_SKILLS_DIR", home / ".deepctl" / "skills") + monkeypatch.setattr(sg, "_STATE_FILE", home / ".deepctl" / "skills" / "skills.json") + for con in (output.console, output.stderr_console, plugin_module.console): + monkeypatch.setattr(con, "_width", 400) + monkeypatch.delenv(skill_bundle.REF_ENV_VAR, raising=False) + monkeypatch.setattr(shutil, "which", lambda name: None) + + def fetch(ref=None): + skills = [] + for name in NAMES: + folder = tmp_path / "bundle" / "skills" / name + folder.mkdir(parents=True, exist_ok=True) + (folder / "SKILL.md").write_bytes(f"---\nname: {name}\n---\n".encode()) + skills.append(RepoSkill(name, folder)) + return skills + + monkeypatch.setattr(skill_bundle, "fetch_skill_bundle", fetch) + saved = dict(output._output_config) + yield home + output._output_config.clear() + output._output_config.update(saved) + + +def seed(): + paths = [] + for n in NAMES: + path = Path.home() / ".claude" / "commands" / "deepgram" / f"{n}.md" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes((FIX / f"{n}.md").read_bytes()) + paths.append(str(path)) + edited = Path.home() / ".cursor" / "rules" / "deepctl.mdc" + edited.parent.mkdir(parents=True) + edited.write_bytes((FIX / "deepctl.mdc").read_bytes() + b"mine\n") + legacy = {"claude": {"paths": paths}, "cursor": {"paths": [str(edited)]}} + sg._STATE_FILE.parent.mkdir(parents=True, exist_ok=True) + sg._STATE_FILE.write_text(json.dumps({"installed_skills": legacy}), "utf-8") + return [Path(p) for p in paths], edited + + +@pytest.mark.parametrize("agentic", [False, True]) +def test_plugin_remove_json_keeps_cleanup_off_stdout(agentic): + output._output_config.update(agentic=agentic, format="json", quiet=False) + files, edited = seed() + cmd = PluginCommand() + ok = PluginOperationResult( + success=True, action="remove", package="foo", message="Removed foo" + ) + group = click.Group("plugin", commands=cmd.setup_commands()) + obj = {"config": MagicMock(), "auth_manager": MagicMock(), "client": MagicMock()} + with patch.object(cmd, "remove_plugin", return_value=ok): + result = CliRunner().invoke(group, ["remove", "foo", "--yes"], obj=obj) + assert result.exit_code == 0, result.output + assert not any(p.exists() for p in files) + assert edited.exists() + assert "0.3.x" not in result.stdout + assert "deepctl can't prove it wrote" not in result.stdout + err = " ".join(result.stderr.split()) + assert "Removed deepctl 0.3.x files for Claude Code" in err + assert "deepctl can't prove it wrote" in err + if agentic: + assert result.stdout == "" diff --git a/packages/deepctl-cmd-skills/src/deepctl_cmd_skills/command.py b/packages/deepctl-cmd-skills/src/deepctl_cmd_skills/command.py index b334b648..28363e0f 100644 --- a/packages/deepctl-cmd-skills/src/deepctl_cmd_skills/command.py +++ b/packages/deepctl-cmd-skills/src/deepctl_cmd_skills/command.py @@ -281,13 +281,14 @@ def _handle_status(self) -> None: notes += [sg._msg("E16", path=p) for p in st.leftovers] if found and st.root is None: notes.append(sg._msg("E15", gen)) - if st.root and gen.cli_name in legacy and gen.cli_name not in recs: + v03 = recs.get(gen.cli_name, {"v03": 1}).get("v03") + if st.root and gen.cli_name in legacy and v03: old.append(gen.display_name) console.print(table) for note in dict.fromkeys(notes): print_warning(escape(note)) if old: - note = f"Files from deepctl 0.3.x are recorded for {', '.join(old)}; run 'dg skills update' to install the skill folders, and the old files stay until a later release." + note = f"Files from deepctl 0.3.x are recorded for {', '.join(old)}; 'dg skills install' or 'dg skills update' removes the ones deepctl can prove it wrote once the skill folders are installed." print_info(escape(note)) if detected and not recs and not legacy: print_info("Run 'dg skills install' to set up AI assistant integrations.") @@ -424,13 +425,17 @@ def _handle_remove( for note in notes: print_warning(escape(note)) paths = legacy.get(cli_key, {}).get("paths", []) - old = recs.get(cli_key, {}).get("v03") or any( - Path(p).parent != gen.skills_root() for p in paths - ) # Not 0.3.x if every path is one of our folders. - v03 = "files from deepctl 0.3.x stay until a later release." + left = [p for p in paths if Path(p).parent != gen.skills_root()] + old = [p for p in left if not sg._v03_gone(p)] # On disk, else no note. + v03 = f"its deepctl 0.3.x files were kept: {', '.join(old)}; delete any you don't need, or 'dg skills install' removes the ones deepctl can prove it wrote." + if cli_key in sg._V03_SHARED: # A section in the user's own file. + v03 = f"the deepctl 0.3.x section in {', '.join(old)}, if any, was kept; 'dg skills install' removes it when it can do so safely; if it is still there afterwards, remove the lines between its marker lines yourself." c10 = f"For {gen.display_name}, {v03}" if cli_key not in recs: c10 = f"{gen.display_name} has no skill folders recorded, so nothing was removed{'; ' + v03 if old else '.'}" + if gen.skills_root() is None and old: + a = " and its entry under 'read:' in ~/.aider.conf.yml" + c10 = f"{gen.display_name} has no skill folders, so nothing was removed and its deepctl 0.3.x file is kept; delete it{a * (cli_key == 'aider')} yourself if you don't need it." if old or cli_key not in recs: print_info(escape(c10)) failed = failed or bool( diff --git a/packages/deepctl-cmd-skills/tests/unit/test_skills_command.py b/packages/deepctl-cmd-skills/tests/unit/test_skills_command.py index 5b953d7f..2cd79c1c 100644 --- a/packages/deepctl-cmd-skills/tests/unit/test_skills_command.py +++ b/packages/deepctl-cmd-skills/tests/unit/test_skills_command.py @@ -441,6 +441,9 @@ def test_remove_exit_one_when_staging_left(self, bundle, monkeypatch, capsys): @pytest.mark.parametrize("old", [True, False]) def test_remove_notes_03x_files_left_behind(self, bundle, capsys, old): detect("claude") + kept = Path.home() / ".claude" / "commands" / "deepgram" / "setup-mcp.md" + kept.parent.mkdir(parents=True) + kept.write_bytes(b"mine") # A folder for setup-mcp never lands: kept. if old: write_state( { @@ -452,7 +455,7 @@ def test_remove_notes_03x_files_left_behind(self, bundle, capsys, old): / ".claude" / "commands" / "deepgram" - / "api.md" + / "setup-mcp.md" ) ] } @@ -464,9 +467,10 @@ def test_remove_notes_03x_files_left_behind(self, bundle, capsys, old): capsys.readouterr() SkillsCommand()._handle_remove(remove_all=True) err = err_text(capsys) + old = old and os.name != "nt" assert ("0.3.x" in err) == old assert ( - "For Claude Code, files from deepctl 0.3.x stay until a later release." + f"For Claude Code, its deepctl 0.3.x files were kept: {kept}; delete any you don't need, or 'dg skills install' removes the ones deepctl can prove it wrote." in err ) == old assert "OK: Removed 2 skill folders from 1 tool." in err @@ -681,8 +685,12 @@ def test_remove_moved_folder_reports_e4_and_exits_one(self, monkeypatch, capsys) def test_remove_notes_03x_files_after_a_plugin_refresh(self, bundle, capsys): detect("claude") - old = Path.home() / ".claude" / "commands" / "deepgram" / "api.md" + old = Path.home() / ".claude" / "commands" / "deepgram" / "setup-mcp.md" + old.parent.mkdir(parents=True) + old.write_bytes(b"mine") # A folder for setup-mcp never lands: kept. rule = Path.home() / ".amazonq" / "rules" / "deepctl.md" + rule.parent.mkdir(parents=True) + rule.write_bytes(b"0.3.x rules") # On disk, so remove names it. write_state( { "installed_skills": { @@ -695,22 +703,24 @@ def test_remove_notes_03x_files_after_a_plugin_refresh(self, bundle, capsys): capsys.readouterr() PluginCommand()._maybe_update_skills() # The real plugin refresh. - assert err_text(capsys) == "" + refresh = err_text(capsys) + if os.name == "nt": + assert "does not clean 0.3.x content on Windows" in refresh + assert str(old) not in disk_state()["installed_skills"]["claude"]["paths"] + else: + assert refresh == "" assert "claude" in disk_state()["skill_folders"] assert disk_state()["installed_skills"]["amazonq"] == {"paths": [str(rule)]} - assert disk_state()["installed_skills"]["claude"]["paths"] == [str(old)] + if os.name != "nt": + assert disk_state()["installed_skills"]["claude"]["paths"] == [str(old)] SkillsCommand()._handle_update() # A later write keeps the flag. capsys.readouterr() SkillsCommand()._handle_remove(remove_all=True) - err, v03 = ( - err_text(capsys), - "files from deepctl 0.3.x stay until a later release.", - ) - assert f"For Claude Code, {v03}" in err - assert ( - f"Amazon Q Developer has no skill folders recorded, so nothing was removed; {v03}" - in err - ) + err = err_text(capsys) + v03 = f"its deepctl 0.3.x files were kept: {old}; delete any you don't need, or 'dg skills install' removes the ones deepctl can prove it wrote." + assert (f"For Claude Code, {v03}" in err) != (os.name == "nt") + q = "Amazon Q Developer has no skill folders, so nothing was removed and its deepctl 0.3.x file is kept; delete it yourself if you don't need it." + assert q in err def test_update_edited_folder_exits_one_with_rename_advice(self, bundle): detect("claude") @@ -812,7 +822,11 @@ def fingerprinted(): assert unchanged() fingerprinted() assert len(disk_state()["skill_folders"]) == 6 - assert disk_state()["installed_skills"] == legacy + recorded = disk_state()["installed_skills"] + assert set(recorded) == set(tools) # Login and startup key on these. + assert {t: recorded[t] for t in ("amazonq", "aider")} == { + t: legacy[t] for t in ("amazonq", "aider") + } # Hint-only tools: never cleaned up. cmd._handle_update() assert unchanged() fingerprinted() diff --git a/packages/deepctl-cmd-skills/tests/unit/test_skills_legacy_v03.py b/packages/deepctl-cmd-skills/tests/unit/test_skills_legacy_v03.py new file mode 100644 index 00000000..34409e8b --- /dev/null +++ b/packages/deepctl-cmd-skills/tests/unit/test_skills_legacy_v03.py @@ -0,0 +1,156 @@ +"""dg skills and the deepctl 0.3.x cleanup: what the commands print around it.""" + +import json +import os +import shutil +from pathlib import Path + +import click +import pytest +from deepctl_cmd_skills import command +from deepctl_cmd_skills.command import SkillsCommand +from deepctl_core import output, skill_bundle +from deepctl_core import skill_generator as sg +from deepctl_core.skill_bundle import RepoSkill + +FIX = Path(__file__).parents[3] / "deepctl-core" / "tests" / "unit" / "fixtures" +FIX = FIX / "legacy_v03" +NAMES = ("api", "docs", "setup-mcp", "starters") +BLOB = {n: (FIX / f"{n}.md").read_bytes() for n in NAMES} +pytestmark = pytest.mark.skipif( + os.name == "nt", reason="legacy cleanup is intentionally disabled on Windows" +) + + +@pytest.fixture(autouse=True) +def home(tmp_path, monkeypatch): + home = tmp_path / "home" + home.mkdir() + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.setenv("HOME", str(home)) + monkeypatch.setenv("USERPROFILE", str(home)) + monkeypatch.setattr(sg, "_SKILLS_DIR", home / ".deepctl" / "skills") + monkeypatch.setattr(sg, "_STATE_FILE", home / ".deepctl" / "skills" / "skills.json") + for con in (output.console, output.stderr_console, command.console): + monkeypatch.setattr(con, "_width", 400) + monkeypatch.delenv(skill_bundle.REF_ENV_VAR, raising=False) + monkeypatch.setattr(shutil, "which", lambda name: None) + saved = dict(output._output_config) + output._output_config.update(agentic=True, format="default", quiet=False) + yield home + output._output_config.clear() + output._output_config.update(saved) + + +def use_bundle(tmp_path, monkeypatch, names=NAMES): + def fetch(ref=None): + skills = [] + for name in names: + folder = tmp_path / "bundle" / "skills" / name + folder.mkdir(parents=True, exist_ok=True) + (folder / "SKILL.md").write_bytes(f"---\nname: {name}\n---\n".encode()) + skills.append(RepoSkill(name, folder)) + return skills + + monkeypatch.setattr(skill_bundle, "fetch_skill_bundle", fetch) + + +def claude(name): + return Path.home() / ".claude" / "commands" / "deepgram" / f"{name}.md" + + +def seed(names=NAMES, extra=None): + Path.home().joinpath(".claude").mkdir(exist_ok=True) + for n in names: + claude(n).parent.mkdir(parents=True, exist_ok=True) + claude(n).write_bytes(BLOB[n]) + legacy = {"claude": {"paths": [str(claude(n)) for n in names]}, **(extra or {})} + sg._STATE_FILE.parent.mkdir(parents=True, exist_ok=True) + sg._STATE_FILE.write_text(json.dumps({"installed_skills": legacy}), "utf-8") + + +def err_text(capsys): + return " ".join(click.unstyle(capsys.readouterr().err).split()) + + +def test_install_removes_proven_files_and_remove_has_no_03x_note( + tmp_path, monkeypatch, capsys +): + use_bundle(tmp_path, monkeypatch) + seed() + SkillsCommand()._handle_install(install_all=True) + err = err_text(capsys) + assert "INFO: Removed deepctl 0.3.x files for Claude Code:" in err + backups = [ + n + for n in claude("api").parent.iterdir() + if n.name.startswith(".deepctl-kept-v03-") + ] + assert len(backups) == len(NAMES) + SkillsCommand()._handle_status() + assert "0.3.x" not in err_text(capsys) + SkillsCommand()._handle_remove(remove_all=True) + assert "0.3.x" not in err_text(capsys) + + +@pytest.mark.parametrize("gone", [False, True]) +def test_kept_file_status_and_remove_name_install_and_update( + tmp_path, monkeypatch, capsys, gone +): + use_bundle(tmp_path, monkeypatch, ("api", "docs", "starters")) + seed() # No setup-mcp folder lands, so setup-mcp.md stays tracked. + SkillsCommand()._handle_install(install_all=True) + capsys.readouterr() + assert claude("setup-mcp").read_bytes() == BLOB["setup-mcp"] + assert not claude("api").exists() + SkillsCommand()._handle_status() + note = "Files from deepctl 0.3.x are recorded for Claude Code; 'dg skills install' or 'dg skills update' removes the ones deepctl can prove it wrote once the skill folders are installed." + assert note in err_text(capsys) + if gone: # Deleted by hand since: nothing to say about 0.3.x files. + claude("setup-mcp").unlink() + SkillsCommand()._handle_remove(remove_all=True) + err = err_text(capsys) + kept = f"For Claude Code, its deepctl 0.3.x files were kept: {claude('setup-mcp')}; delete any you don't need, or 'dg skills install' removes the ones deepctl can prove it wrote." + assert (kept in err) != gone + assert ("0.3.x" in err) != gone + + +def test_kept_shared_section_is_not_called_a_deepctl_file( + tmp_path, monkeypatch, capsys +): + use_bundle(tmp_path, monkeypatch, ("api", "docs", "starters")) + gem = Path.home() / ".gemini" / "GEMINI.md" + gem.parent.mkdir(parents=True) + gem.write_bytes(b"my notes\n\n" + (FIX / "GEMINI.md").read_bytes()) + before = gem.read_bytes() + seed(names=(), extra={"gemini": {"paths": [str(gem)]}}) + SkillsCommand()._handle_install(install_all=True) # No setup-mcp: kept. + capsys.readouterr() + SkillsCommand()._handle_remove(remove_all=True) + err = err_text(capsys) + note = f"For Gemini CLI, the deepctl 0.3.x section in {gem}, if any, was kept; 'dg skills install' removes it when it can do so safely; if it is still there afterwards, remove the lines between its marker lines yourself." + assert note in err + assert "files were kept" not in err and "delete any" not in err + assert gem.read_bytes() == before + + +def test_hint_only_remove_says_the_file_is_kept(tmp_path, monkeypatch, capsys): + use_bundle(tmp_path, monkeypatch) + rule = Path.home() / ".amazonq" / "rules" / "deepctl.md" + rule.parent.mkdir(parents=True) + rule.write_bytes(b"0.3.x rules") + conv = Path.home() / ".deepctl" / "skills" / "deepctl-conventions.md" + conv.parent.mkdir(parents=True) + conv.write_bytes(b"0.3.x conventions") # On disk, so remove names it. + seed(extra={"amazonq": {"paths": [str(rule)]}, "aider": {"paths": [str(conv)]}}) + SkillsCommand()._handle_install(install_all=True) + capsys.readouterr() + SkillsCommand()._handle_remove(remove_all=True) + text = err_text(capsys) + assert ( + f"INFO: Amazon Q Developer has no skill folders, so nothing was removed and its deepctl 0.3.x file is kept; delete it yourself if you don't need it." + in text + ) + aider = "Aider has no skill folders, so nothing was removed and its deepctl 0.3.x file is kept; delete it and its entry under 'read:' in ~/.aider.conf.yml yourself if you don't need it." + assert aider in text + assert rule.read_bytes() == b"0.3.x rules" diff --git a/packages/deepctl-core/src/deepctl_core/output.py b/packages/deepctl-core/src/deepctl_core/output.py index 54d5a3d7..8888245a 100644 --- a/packages/deepctl-core/src/deepctl_core/output.py +++ b/packages/deepctl-core/src/deepctl_core/output.py @@ -380,13 +380,13 @@ def print_warning(message: str, *, stderr: bool = False) -> None: ) -def print_info(message: str) -> None: +def print_info(message: str, *, stderr: bool = False) -> None: """Print info message.""" if not _output_config["quiet"]: if _output_config["agentic"]: stderr_console.print(f"INFO: {message}") else: - console.print(f"[blue]ℹ[/blue] {message}") + (stderr_console if stderr else console).print(f"[blue]ℹ[/blue] {message}") def print_debug(message: str) -> None: diff --git a/packages/deepctl-core/src/deepctl_core/skill_generator.py b/packages/deepctl-core/src/deepctl_core/skill_generator.py index c6f4afa0..44f8e94d 100644 --- a/packages/deepctl-core/src/deepctl_core/skill_generator.py +++ b/packages/deepctl-core/src/deepctl_core/skill_generator.py @@ -33,7 +33,7 @@ from rich.markup import escape from deepctl_core import skill_bundle -from deepctl_core.output import print_warning +from deepctl_core.output import _output_config, print_info, print_warning from deepctl_core.skill_bundle import portable_name if sys.platform == "win32": @@ -76,6 +76,7 @@ {"i386": 353, "i686": 353, "armv7l": 382, "armv6l": 382, "arm": 382}, ) +_V03_LINES = "only the lines from '' yourself and keep the rest of the file" # In E34 and E40. # One sentence each. {file} is skills.json; {display} and {root} name the tool. _MSG = { "E1": "deepctl cannot prove it installed {paths}, so it will not replace anything there; move or rename what is there, then run the command again.", @@ -99,6 +100,19 @@ "E16": "{path} looks like staging from an interrupted deepctl run; deepctl never deletes it, so check it and delete it by hand.", "E18": "{root} exists but is not a folder, so deepctl changed nothing for {display}; move it away or point it at a folder, then run the command again.", "E21": "Could not remove the skills from {root}: {reason}.", + "E33": "deepctl can't prove it wrote {path} ({why}), so it left it in place and no longer tracks it; if it's an old deepctl 0.3.x copy you don't need, delete it.", + "E34": "deepctl can't safely remove its 0.3.x section from {path} ({why}), so it left the file as it is and won't warn about it again; a later install or update removes the section once it can, or remove {what}.", + "E35": "Could not remove deepctl 0.3.x content from {path}: {reason}; {path} is unchanged and still recorded, so the next install or update tries again.", + "E35b": "Could not remove deepctl 0.3.x content from {path}: {reason}; it is still recorded, so the next install or update tries again.", + "E36": "The skills for {display} are installed, but deepctl could not finish removing its 0.3.x files: {reason}; the next install or update tries again.", + "E37": "{dest} was saved while deepctl was removing its 0.3.x content, so deepctl kept your save; the earlier version is in {aside}. Compare them before you delete {aside}.", + "E38": "{aside}, left by an earlier deepctl run, holds an earlier version of {dest}, so deepctl changed neither; compare them, keep what you want in {dest}, then delete {aside}.", + "E39": "{dest} was missing, so deepctl put it back from {aside}, where an interrupted deepctl run had moved it.", + "E40": "{why.strerror}, so deepctl left {path} as it is and won't warn about it again; if it holds deepctl 0.3.x content you don't need, remove {what}.", + "E41": "{aside} is no longer the file deepctl moved there (a link or folder is there now), so deepctl did not put it back and {dest} is missing; restore {dest} from a backup if you need it, then delete {aside}.", + "E42": "{aside} is not a file deepctl moved there, so deepctl changed neither it nor {dest}; delete {aside} if you don't need it.", + "E43": "deepctl removed its 0.3.x content from {dest}, but kept the original file in {aside} because a program that already had it open can still write it; compare it, then delete {aside} if you don't need it.", + "E44": "deepctl does not clean 0.3.x content on Windows because a parent folder can become a junction during cleanup; it left these recorded paths untouched and will not try again: {paths}. Delete the content you don't need yourself.", "E22": "{dest} was edited since deepctl installed it, so deepctl left it alone and did not install over it; rename or move your edited folder, then run the command again.", "E23": "{dest} was edited since deepctl installed it, so deepctl left it in place and no longer tracks it; delete it yourself if you don't need it.", "E24": "{dest} was edited since deepctl installed it, so 'dg skills remove' leaves it alone and 'dg skills update' stops until you rename or move it to keep your edits, or delete it to get deepctl's copy back.", @@ -293,13 +307,13 @@ def _marker_text(cli: str, name: str) -> str: return f"deepctl installed this folder ({cli}/{name}); 'dg skills update' replaces it and 'dg skills remove' deletes it.\n" -def _read_regular(path: str | Path, limit: int) -> bytes | None: +def _read_regular(path: str | Path, limit: int, at: int | None = None) -> bytes | None: """Read one regular file of at most ``limit`` bytes, never via a link, else None.""" - lst = os.lstat(path) # The link check on Windows, which has no O_NOFOLLOW. + lst = os.stat(path, dir_fd=at, follow_symlinks=False) # Windows: no O_NOFOLLOW. if _is_link(lst) or not stat.S_ISREG(lst.st_mode) or lst.st_size > limit: return None flags = os.O_RDONLY | getattr(os, "O_BINARY", 0) | getattr(os, "O_NOFOLLOW", 0) - fd = os.open(path, flags | getattr(os, "O_NONBLOCK", 0)) + fd = os.open(path, flags | getattr(os, "O_NONBLOCK", 0), dir_fd=at) try: st, data = os.fstat(fd), bytearray() if (st.st_dev, st.st_ino) != (lst.st_dev, lst.st_ino) or st.st_size > limit: @@ -367,20 +381,23 @@ def _ownership(path: Path, cli: str, name: str, rec: dict[str, Any] | None) -> s return "ok" if fp in want else "edited" -def _rename_excl(src: Path, dest: Path) -> None: +def _rename_excl(src: str | Path, dest: str | Path, at: int | None = None) -> None: """Rename ``src`` to ``dest`` in one step that fails if anything is at ``dest``.""" if _WINDOWS: os.rename(src, dest) # Windows rename refuses any existing dest. return mac, libc = sys.platform == "darwin", ctypes.CDLL(None, use_errno=True) fn = getattr(libc, "renamex_np" if mac else "renameat2", None) + if mac and at is not None: # Names relative to folder fd ``at`` (macOS 10.12+). + fn = getattr(libc, "renameatx_np", None) nr = _NR_RENAMEAT2[sys.maxsize < 2**32].get(platform.machine()) if nr and sys.platform == "linux" and hasattr(libc, "syscall"): # Every glibc. fn = functools.partial(libc.syscall, ctypes.c_long(nr)) # The kernel's errno. if fn is None: # No call to make on this OS, machine or Python. raise OSError(errno.ENOSYS, _NO_EXCL_SYS, str(dest)) a, b = os.fsencode(src), os.fsencode(dest) - if (fn(a, b, 4) if mac else fn(-100, a, -100, b, 1)) != 0: # RENAME_EXCL/NOREPLACE + d = -100 if at is None else at # AT_FDCWD (Linux); RENAME_EXCL 4, NOREPLACE 1. + if (fn(a, b, 4) if mac and at is None else fn(d, a, d, b, 4 if mac else 1)) != 0: e = ctypes.get_errno() or errno.EIO # Never "Success" for a failed call. bad = e in (errno.EINVAL, errno.ENOTSUP, errno.EOPNOTSUPP) # The filesystem. why = _NO_EXCL_SYS if e == errno.ENOSYS else _NO_EXCL if bad else None @@ -392,6 +409,359 @@ def _place(src: Path, dest: Path) -> None: _rename_excl(src, dest) +# "bytes:sha256" of each SKILL.md 0.2.16-0.3.2 copied (legacy_v03/allowlist.tsv). +_V03_BLOBS = { + "api": ( + "2271:0e84ca7cdbfecde6ccbad869ac1368ffc70e500d7fd63cba317ef5c5a010bc39", + "6892:1e3c33188e3b6548adac918916e489cecc9dd408cc63eccc7645846a9bf8b5ef", + "7229:37b0c83a184100354b58dbd6f63fe086e11018aec0e29730a58a71cb72ba8569", + "7810:87682eb16a5fe904bad30ee1f68c43dc1a6db31217c252e1d2b94e89c8c810ac", + "7558:ab6dcec901dbe89994ee8b5f43649d488fca95d3ad591d910f6109b5946fea97", + "7769:644f06c0a29a2251a556c5669d26f636074ebfb1ab57f61d7c298d34f5810553", + "11915:db3f40de8edb8b810ec9636cf2d5ac8916a4c760555373d5b4f4a75607b9ec59", + "12087:523e206af4c33a07175d7fd6d190b70ed7b1c7ec89cb3b6b4575669abf02e5a2", + "19480:b2855ce6bcc9d8e6744c9b669c8ad0de624100f779a80d9139b73333bfd5e408", + "19496:b193fe2baed574077026cf2b60f5d07985ad27899e2e8ffdab6a2657d70601a6", + "21751:8cda50a65b00eb45b3789fd3e29bc998f7737aa0ad03bb563e782cc3924b556b", + "25667:545d78a1b2735479237fe7703128ca033279126772eca2a6dbf1a5a878ea12f2", + "26114:f24514384d9662214b973923117802bf5b7329be17b787be0ed5495cb668eab1", + "29407:959031e436ae0eb50bb5139a30acca1607c3f4828061e655b6b7e3d10be1ea92", + ), + "docs": ( + "1683:a3d3c853b73b6e0f56cba86a6be91bcba936135d4a58fa0ecc62958de1349e12", + "3511:e987e38d0e832c11949a21395c38cec2a0e7c275acfce26e65a0c7c36cffaa0b", + "4875:cef47147b79e903c72b3d27a9bd8dcca3ddeb5e0f9f6bbc36da4dd6f13ad8336", + "4925:1c0457b580de0620b8953bb6029873d614eb96e586b7e6be72c83fd63c2e51a8", + "5249:64e23bb3edd797bc149a3fd26034170e1c8e2850d608ad15e59915bfa84f289e", + ), + "setup-mcp": ( + "4647:8ce952a6d4322ea883028c9a548be58fc7178bbba4148baae257aa57d3a7d68a", + "10674:251db18b9a887660093e82a09b4c3ff02d464e02df1450484a87261d2ce836b3", + "12301:f5000298802362356907889bfcd90cba430c528d466c3f27bddd3923274d9a54", + ), + "starters": ( + "8790:9ba321bc8cf444c8b493d290d61e5dda00fb21bb7a56b55bef7b92952a841b2c", + "10027:6a5622355f3c185b2eafd3dad5b54aca01bda47d11d03dff79f6ab4228194c70", + "11464:0074b2b8f7677624ee0a7f94d373085ea70c87e5057b601adf549094e9f61258", + "11489:b92ba4683fc740b77858a3f7b2f9845b6ce07fee91d17ba7ec67d1417b4c70cb", + "16472:40c3ec366c632145a619276fee54f426a124a496ebcfcee4b1fa0f34d7a25b9b", + "16572:950c42c6c7c01f5692a5a2cdc00c6e1bbded51dab0e261e290f890f8eda035a0", + "17376:aba86630c4872031d3c66dc100e58b3878a3c9b4cbbfad4af87a12102ba8e228", + ), +} +_V03_BEGIN = b"" +_V03_END = b"" +_V03_SEP = b"\n\n---\n\n" # 0.3.x joined the skills with this, in _V03_BLOBS order. +_V03_ASIDE = ".deepctl-v03-" # Not _STAGING_PREFIX: README names 0.3.x leftovers. +_V03_MAX = 16 << 20 +_V03_DIR = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) +_V03_NOFOLLOW = _V03_DIR | getattr(os, "O_NOFOLLOW", 0) +_V03_NEW = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_BINARY", 0) +_V03_SHARED = ("codex", "gemini", "opencode") # A section in the user's own file. +_V03_CLAUDE = ".claude/commands/deepgram" +_V03_PATHS = { # Under the home folder, from the v0.3.2 generator. + "claude": [f"{_V03_CLAUDE}/{n}.md" for n in _V03_BLOBS], + "cursor": [".cursor/rules/deepctl.mdc"], + "cline": [".cline/rules/deepctl.md"], + "codex": [".codex/instructions.md"], + "gemini": [".gemini/GEMINI.md"], + "opencode": [".opencode/agents.md"], +} + + +def _v03_join(data: bytes, names: list[str]) -> bool: + """True if ``data`` is allowlisted blobs, one per name at most, in order, joined.""" + for i, n in enumerate(names): + for blob in _V03_BLOBS[n]: + size, sha = blob.split(":") + head, rest = data[: int(size)], data[int(size) :] + tail = rest[len(_V03_SEP) :] if rest.startswith(_V03_SEP) else None + if hashlib.sha256(head).hexdigest() == sha and ( + not rest or (tail is not None and _v03_join(tail, names[i + 1 :])) + ): + return True + return False + + +def _v03_gone(path: str | Path) -> bool: + """True only when ``path`` provably does not exist; an unreadable one is there.""" + try: + return not os.lstat(path) # A stat result is never empty: it is there. + except OSError as exc: + return isinstance(exc, (FileNotFoundError, NotADirectoryError)) + + +class _V03Link(OSError): ... # A folder between home and a legacy file is a link. + + +@dataclass(frozen=True) +class _V03Done: + """A finished cleanup and the original inode retained for late writers.""" + + removed: bool + aside: Path + + +class _V03Dir(contextlib.AbstractContextManager["_V03Dir"]): + """Folder of ``rel``, reached through no link: by fd (POSIX), rechecked (Windows).""" + + def __init__(self, rel: str) -> None: + *self.parts, self.name = rel.split("/") + self.where = Path.home().joinpath(*self.parts) + self.walk() + + def walk(self, err: type[OSError] = _V03Link) -> None: + home = Path.home() # Followed: HOME itself may be a link (/home -> /data/home). + self.fd: int | None = None if _WINDOWS else os.open(home, _V03_DIR) + try: + for i, part in enumerate(self.parts): + if _is_link(os.lstat(sub := home.joinpath(*self.parts[: i + 1]))): + why = f"{sub} is a link, which deepctl doesn't follow" + raise err(errno.ELOOP, why, str(sub)) + if self.fd is not None: # A link swapped in since the lstat fails here. + up, self.fd = self.fd, os.open(part, _V03_NOFOLLOW, dir_fd=self.fd) + os.close(up) + except BaseException: + self.__exit__() + raise + + def __call__(self, name: str) -> str: + if self.fd is not None: + return name # Used with dir_fd=self.fd. + self.walk(OSError) # Not E40: a file may have moved already (E35, tracked). + return str(self.where / name) + + def lstat(self, name: str) -> os.stat_result | None: + try: + return os.stat(self(name), dir_fd=self.fd, follow_symlinks=False) + except FileNotFoundError: + return None + + def back(self, aside: str) -> bool: + """Put ``aside`` back (never over a newer file); False after E4, E37 or E41.""" + try: + if (s := self.lstat(aside)) and not stat.S_ISREG(s.st_mode): + return self.warn("E41", aside) # A link or folder is there now. + if s: # Nothing if nothing moved. + _rename_excl(self(aside), self(self.name), self.fd) + return True + except OSError as exc: # A save at the name since (E37), else E4. + return self.warn("E37" if exc.errno in _NO_REPLACE else "E4", aside) + + def warn(self, key: str, aside: str) -> bool: + text = _msg(key, dest=self.where / self.name, aside=self.where / aside) + print_warning(escape(text), stderr=True) + return False # For back(): the file is not back. + + def __exit__(self, *exc: object) -> None: + if self.fd is not None: + os.close(self.fd) + + +def _v03_mv( + d: _V03Dir, data: bytes, st: os.stat_result, new: bytes | None, why: str +) -> _V03Done | str: + """After a re-proof, delete or publish ``new`` while retaining the original inode.""" + aside, tmp, pub = _V03_ASIDE + d.name, f"{_V03_ASIDE}{uuid.uuid4().hex}.tmp", False + kept = f".deepctl-kept-v03-{uuid.uuid4().hex}-{d.name}" + made: tuple[int, ...] | None = None + try: + if new is not None: + with os.fdopen(os.open(d(tmp), _V03_NEW, 0o600, dir_fd=d.fd), "wb") as f: + f.write(new) + f.flush() + os.fsync(f.fileno()) + at = d(tmp) if _WINDOWS else f.fileno() # POSIX: by fd, never a link. + os.chmod(at, stat.S_IMODE(st.st_mode)) # Like copystat. + made = os.fstat(f.fileno())[:3] # Its type and mode, inode, device. + if not _WINDOWS: + os.utime(at, ns=(st.st_atime_ns, st.st_mtime_ns)) + if _WINDOWS: # After the close, which would reset the time there. + os.utime(at, ns=(st.st_atime_ns, st.st_mtime_ns)) + try: # Ctrl-C right after the move still puts it back. + _rename_excl(d(d.name), d(aside), d.fd) # A save from now on lands at name. + if (bad := _read_regular(d(aside), _V03_MAX, d.fd) != data) or made != ( + (s := d.lstat(tmp)) and s[:3] # The temp it wrote, not a link; a delete + ): # has no temp, so None == None there. + why = why if bad else "deepctl's temporary copy of it was replaced" + return why if d.back(aside) else "" # E4, E37 or E41 said where it is. + if new is None: + _rename_excl(d(aside), d(kept), d.fd) + return _V03Done(True, d.where / kept) + pub = True + _rename_excl(d(tmp), d(d.name), d.fd) # Refused if a new file is there. + except BaseException as exc: + if pub and isinstance(exc, OSError) and exc.errno in _NO_REPLACE: + d.warn("E37", aside) + return "" + left = True # Unless it was published: the next run names it (E38). + with contextlib.suppress(OSError): # A link now: Ctrl-C still re-raises. + left = not pub or isinstance(exc, OSError) or bool(d.lstat(tmp)) + vars(exc)["v03_moved"] = left and not d.back(aside) # E35b, not E35. + raise + _rename_excl(d(aside), d(kept), d.fd) + return _V03Done(False, d.where / kept) + finally: + with contextlib.suppress(OSError): + os.unlink(d(tmp), dir_fd=d.fd) # Gone already once it was published. + + +def _v03_file(rel: str, names: list[str], shared: bool) -> _V03Done | str | _V03Link: + """A completed cleanup, no-op, or reason kept; raises OSError if I/O fails.""" + try: + with _V03Dir(rel) as d: + return _v03_cut(d, names, shared) + except _V03Link as exc: + return exc + except (FileNotFoundError, NotADirectoryError): + return "" + + +def _v03_cut(d: _V03Dir, names: list[str], shared: bool) -> _V03Done | str: + aside = _V03_ASIDE + d.name # The same name each run, so a leftover is found. + st, old = d.lstat(d.name), d.lstat(aside) + if old and st: # E38 for a file an earlier run moved; else not deepctl's (E42). + d.warn("E38" if stat.S_ISREG(old.st_mode) else "E42", aside) + return "" + if old and stat.S_ISREG(old.st_mode) and not _is_link(old): # An interrupted run's. + try: + _rename_excl(d(aside), d(d.name), d.fd) # Refused if a file appeared since. + except OSError as exc: + vars(exc)["v03_moved"] = True # E35b: it is still in the aside. + raise + d.warn("E39", aside) + st = old + if st is None: + return "" + data = _read_regular(d(d.name), _V03_MAX, d.fd) + if data is None: + if _is_link(st) or not stat.S_ISREG(st.st_mode): + return "it is a link" if _is_link(st) else "it is not a file" + big = st.st_size > _V03_MAX # Else it changed between the lstat and the read. + return "it is larger than 16 MiB" if big else "it changed while deepctl read it" + eol = b"\r\n" if b"\r\n" in data else b"\n" + mixed = b"\n" in data.replace(eol, b"") # CRLF and LF: 0.3.x never did. + if not shared: + if mixed or not _v03_join(data.replace(b"\r\n", b"\n"), names): + return "it differs from every deepgram/skills version deepctl 0.3.x copied" + return _v03_mv(d, data, st, None, "it changed while deepctl was removing it") + if _V03_BEGIN not in data and _V03_END not in data: + return "" + i, j = data.find(_V03_BEGIN), data.find(_V03_END) + len(_V03_END) + if mixed: + return "it mixes line endings" + if not ( + data.count(_V03_BEGIN) == data.count(_V03_END) == 1 + and i < j + and (i == 0 or data[i - 1 : i] == b"\n") + and data[i + len(_V03_BEGIN) :].startswith(eol) + and data[: j - len(_V03_END)].endswith(eol) + and (j == len(data) or data[j:].startswith(eol)) + ): + return "its deepctl section is incomplete, repeated or not on lines of its own" + if st.st_nlink > 1 or (hasattr(os, "getuid") and st.st_uid != os.getuid()): + return "it has other hard links or another user owns it" + ro = not st.st_mode & 0o222 or not os.access(d(d.name), os.W_OK, dir_fd=d.fd) + if ro or getattr(st, "st_flags", 0) & (stat.UF_IMMUTABLE | stat.SF_IMMUTABLE): + return "it is read-only or locked" + head, tail = data[:i], data[j + len(eol) :] + if head.endswith(eol * 2) and not head[: -2 * len(eol)].endswith(b"\n"): + head = head[: -len(eol)] # The one blank line 0.3.x added before its section. + new = head + tail or None # None: 0.3.x created the file for its section. + return _v03_mv(d, data, st, new, "it changed while deepctl was editing it") + + +def _clean_v03(gen: SkillGenerator, root: Path) -> None: + """Remove the 0.3.x content of ``gen`` deepctl can prove; only Ctrl-C raises.""" + cli, notes, untrack = gen.cli_name, list[str](), set[Path]() + removed, cut, kept, retry = ( + list[Path](), + list[Path](), + list[tuple[Path, Path]](), + set[Path](), + ) + try: + state = get_skills_state() + tool = state.get(_RECORDS_KEY, {}).get(cli, {}) + folders = tool.get("folders", {}) + landed = {n for n, r in folders.items() if r.get("state") == "installed"} + rec = state["installed_skills"].get(cli) + recorded = {Path(p) for p in rec["paths"]} if rec else set() + if _WINDOWS: + legacy = [p for p in recorded if p.parent != root] + if rec and legacy: + paths = [p for p in rec["paths"] if Path(p).parent == root] or [ + str(root / n) for n in sorted(landed) + ] + + def clear_windows(state: dict[str, Any]) -> None: + state["installed_skills"][cli]["paths"] = paths + state.get(_RECORDS_KEY, {}).get(cli, {}).pop("v03", None) + + _update_state(clear_windows, "E9c", gen) + print_warning( + escape(_msg("E44", paths=", ".join(map(str, legacy)))), stderr=True + ) + return + for rel in _V03_PATHS.get(cli, []): + path = Path.home().joinpath(*rel.split("/")) + names = [path.stem] if cli == "claude" else list(_V03_BLOBS) + if not set(names) <= landed: + continue # The folders that replace it did not all land: keep it. + try: + why = _v03_file(rel, names, cli in _V03_SHARED) + except OSError as exc: # Warned only if 0.3.x recorded it (E35 says so). + if path in recorded: # E35b after E4, E37, E41 or a failed E39. + key = "E35b" if getattr(exc, "v03_moved", False) else "E35" + notes.append(_msg(key, path=path, reason=_reason(exc))) + retry.add(path) # Tracked, even if a link now hides it from the prune. + continue + if isinstance(why, _V03Done): + (removed if why.removed else cut).append(path) + kept.append((path, why.aside)) + elif why and path in recorded: # Once, then untracked (as E23/E26). + if _output_config["quiet"]: + continue # Unseen: stay tracked so a later run warns. + key = "E34" if cli in _V03_SHARED else "E33" + key = "E40" if isinstance(why, _V03Link) else key + what = _V03_LINES if cli in _V03_SHARED else "that content yourself" + notes.append(_msg(key, path=path, why=why, what=what)) + untrack.add(path) + if removed and cli == "claude": # Only when empty: the user's files stay. + with contextlib.suppress(OSError), _V03Dir(_V03_CLAUDE) as d: + if (s := d.lstat(d.name)) and not _is_link(s): # Windows junction. + os.rmdir(d(d.name), dir_fd=d.fd) + if rec: + paths = [ + p + for p in rec["paths"] + if Path(p) not in untrack + and (Path(p).parent == root or Path(p) in retry or not _v03_gone(p)) + ] or [str(root / n) for n in sorted(landed)] + v03 = any(Path(p).parent != root for p in paths) + + def clear(state: dict[str, Any]) -> None: + state["installed_skills"][cli]["paths"] = paths + if not v03: + state[_RECORDS_KEY][cli].pop("v03", None) + + if paths != rec["paths"] or (not v03 and "v03" in tool): + _update_state(clear, "E9c", gen) + except Exception as exc: # Never fail an install that landed. + why = str(exc) if isinstance(exc, SkillInstallError) else _reason(exc) + notes.append(_msg("E36", gen, reason=why.rstrip("."))) + if removed: + done = f"Removed deepctl 0.3.x files for {gen.display_name}: {', '.join(map(str, removed))}." + print_info(escape(done), stderr=True) + for p in cut: + text = f"Removed the deepctl 0.3.x section from {p}; the rest of the file is unchanged." + print_info(escape(text), stderr=True) + for path, aside in kept: + print_info(escape(_msg("E43", dest=path, aside=aside)), stderr=True) + for note in notes: + print_warning(escape(note), stderr=True) + + @dataclass(frozen=True) class SkillGenerator: """One AI coding tool and the skills root deepctl installs into.""" @@ -683,6 +1053,7 @@ def settle(state: dict[str, Any]) -> None: # From proof on disk, not ``placed`` error.leftover = leftover if error is not None: raise error + _clean_v03(gen, root) return [root / n for n in placed], leftover diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/GEMINI.md b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/GEMINI.md new file mode 100644 index 00000000..38fd8389 --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/GEMINI.md @@ -0,0 +1,999 @@ + +--- +name: api +description: > + Deepgram API reference for speech-to-text, text-to-speech, voice agents, audio intelligence, + and account management. Use whenever building with Deepgram APIs — REST or WebSocket. Covers + authentication, all endpoints, query parameters, request/response schemas, and WebSocket + message formats. Reference files are organized by domain: listen (STT — Nova and Flux STT), speak + (TTS — Aura and Flux TTS), agent (voice agents), read (text/audio intelligence), models, + projects, auth, and self-hosted. +--- + +# Deepgram API + +Build with Deepgram's speech-to-text, text-to-speech, voice agent, and audio intelligence APIs. + +> **"Flux" names two separate products.** **Flux STT** is conversational speech-to-text on `/v2/listen` (`model=flux-general-en`). **Flux TTS** is turn-based speech synthesis on `/v2/speak` (`model=flux-{voice}-{language}`). They share a name and a design philosophy — turn-aware, built for voice agents — but they are different endpoints with different models, params, and messages. When a request just says "Flux", check whether it is about transcribing audio or producing it. + +## Getting Started + +All API requests require authentication via API key or JWT: + +- **API Key**: `Authorization: Token ` +- **JWT**: `Authorization: Bearer ` + +Base servers: + +- REST & STT/TTS WebSocket: `https://api.deepgram.com` +- Voice Agent WebSocket **and `GET /v1/agent/settings/think/models`**: `https://agent.deepgram.com` + +`GET /v1/agent/settings/think/models` lives on the `agent.` host too, not on `api.`: it +returns 404 on `api.deepgram.com` and 200 on `agent.deepgram.com`. Everything else REST +stays on `api.deepgram.com`. + +### Regional endpoints + +To keep processing inside a geography, swap the host. Same API keys, same paths, same SDKs — +only the base URL changes. Requests are never routed out of region: if the region is +unavailable they fail rather than fall back. + +| Region | Host | +|---|---| +| EU | `api.eu.deepgram.com` | +| Australia | `api.au.deepgram.com` | +| India | `api.in.deepgram.com` | + +**The data plane is regional; the Projects management API is not.** On all three regional hosts: + +| Endpoint | Regional | +|---|---| +| `POST /v1/listen`, `wss://…/v1/listen` | Yes | +| `wss://…/v2/listen` | Yes | +| `POST /v1/speak`, `wss://…/v1/speak` | Yes | +| `POST /v2/speak`, `wss://…/v2/speak` | Yes | +| `POST /v1/read` | Yes | +| `wss://…/v1/agent/converse` | Yes | +| `GET /v1/models` | Yes | +| `POST /v1/auth/grant` | Yes | +| `/v1/projects/*` (keys, members, usage, billing) | **No — 404** | + +Two host rules that catch people out: + +1. **Voice Agent moves onto the `api.` host regionally.** There is no `agent.eu.deepgram.com` + (the name does not resolve). Use `wss://api.eu.deepgram.com/v1/agent/converse`. `GET /v1/agent/settings/think/models` moves with it. Globally it stays on `agent.deepgram.com`. +2. **Keep management calls on `api.deepgram.com`.** Point a client's management calls at a + regional host and `/v1/projects` returns 404, so split the base URL by call type if your + app both transcribes and manages keys. + +Whisper models are not served in any of the three regions — use Nova or Flux STT models there. + +For Deepgram Dedicated and self-hosted hosts, see +[Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints); +for the full per-region feature matrix and SDK snippets, see +[Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints). + +## How Deepgram's APIs Fit Together + +``` + ┌──────────────────────────────┐ + │ api.deepgram.com │ + └──────────────────────────────┘ + │ + ┌───────────┬───────────┬─────┴─────┬───────────┬───────────┐ + ▼ ▼ ▼ ▼ ▼ ▼ + /v1/listen /v2/listen /v1/speak /v2/speak /v1/read /v1/projects/* + Nova — STT Flux — STT Aura — TTS Flux — TTS Text AI Management + REST + WSS WSS only REST + WSS REST + WSS REST only REST only + + ┌──────────────────────────────┐ + │ agent.deepgram.com │ + └──────────────────────────────┘ + │ + ▼ + /v1/agent/converse + WebSocket only + audio ──▶ STT ──▶ LLM ──▶ TTS ──▶ audio + (Deepgram orchestrates the full pipeline) +``` + +## Which API Should I Use? + +``` +Audio → text (transcription)? +├─ General-purpose transcription (captions, batch, call logs, live streams with custom turn logic) +│ └─ Nova models via /v1/listen +│ ├─ Pre-recorded file → REST POST https://api.deepgram.com/v1/listen?model=nova-3 +│ └─ Live stream → WSS wss://api.deepgram.com/v1/listen?model=nova-3 +│ +└─ Conversational audio / voice-agent-style turn detection + └─ Flux STT models via /v2/listen + └─ Live stream → WSS wss://api.deepgram.com/v2/listen?model=flux-general-en + +Text → audio (speech synthesis)? +├─ General-purpose TTS (broadest voice catalog, compressed/containerized audio) +│ └─ Aura models via /v1/speak +│ ├─ One-shot → REST POST https://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ └─ Low-latency stream → WSS wss://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ +└─ Voice-agent TTS (turn-based lifecycle, barge-in, cross-turn consistency) + └─ Flux TTS models via /v2/speak — model is REQUIRED, and must be flux-* + ├─ Pre-render a block → REST POST https://api.deepgram.com/v2/speak?model=flux-alexis-en + └─ Live conversation → WSS wss://api.deepgram.com/v2/speak?model=flux-alexis-en + +Full conversational voice agent (audio in, audio out)? +└─ WSS wss://agent.deepgram.com/v1/agent/converse + Deepgram handles STT + your configured LLM + TTS internally + +Analyze text for insights? +└─ REST POST /v1/read + (summaries, sentiment, topics, intents) +``` + +## Speech-to-Text: Nova (`/v1/listen`) vs Flux STT (`/v2/listen`) + +Both model families are actively maintained and industry-leading. They solve different problems — pick the one that matches your use case. + +| | Nova (`/v1/listen`) | Flux STT (`/v2/listen`) | +|---|---|---| +| Endpoint | `/v1/listen` | `/v2/listen` | +| Available models | `nova-3` (also `nova-3-medical`, `nova-3-pharma`), `nova-2`, `nova`, `enhanced`, `base` | `flux-general-en`, `flux-general-multi` | +| Best for | General transcription — captions, subtitles, call logs, batch | Conversational audio — voice agents, interactive assistants, turn-taking UIs | +| Output | Continuous transcript stream | Structured turn events + transcripts (built-in turn state machine) | +| Turn detection | Manual (`utterance_end_ms`, VAD events) | Built-in (EOT, eager-EOT, turn_index) | +| Transports | REST + WebSocket | WebSocket only | +| Intelligence overlays | Yes — `summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`, etc. | No — smaller focused param set; no `smart_format` / `diarize_model` / `punctuate` | +| Mid-session reconfig | No (reconnect to change) | Yes (`Configure` message updates EOT thresholds, keyterms, language hints, and `numerals` live) | + +**Pick Nova (`/v1/listen`, `model=nova-3`) when:** +- Generating captions, subtitles, or transcripts for recorded media +- Running batch transcription over files (REST) +- You need analytics overlays (`summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`) +- You want WebSocket streaming with your own turn-detection logic + +**Pick Flux STT (`/v2/listen`, `model=flux-general-en`) when:** +- Building an interactive voice agent or assistant +- You want end-of-turn detection handled for you +- You need low-latency turn signals and barge-in support +- You want to update EOT thresholds, keyterms, language hints, or `numerals` mid-session without reconnecting + +Migrating from Nova 3 to Flux STT? See the official [Nova 3 → Flux migration guide](https://developers.deepgram.com/docs/flux/nova-3-migration). + +## Text-to-Speech: Aura (`/v1/speak`) vs Flux TTS (`/v2/speak`) + +Both TTS families are actively maintained. `/v2/speak` is a **new endpoint, not a replacement** — `/v1/speak` is unchanged, and there is no aliasing, redirect, or deprecation. The families do not overlap: Aura voices are served only on `/v1/speak`, Flux TTS voices only on `/v2/speak`. + +| | Aura (`/v1/speak`) | Flux TTS (`/v2/speak`) | +|---|---|---| +| Endpoint | `/v1/speak` | `/v2/speak` | +| Models | `aura-2-*` (en, es, de, nl, fr, it, ja), `aura-*` | `flux-{voice}-{language}`, e.g. `flux-alexis-en` — English at launch | +| `model` param | Optional (defaults to `aura-asteria-en`) | **Required**; an `aura-*` string is rejected | +| Best for | Broadest voice catalog, multilingual, compressed audio, one-shot synthesis | Voice agents — streaming LLM output, barge-in, multi-turn conversations | +| Mental model | Text buffer → audio stream | Streaming-first, turn-based conversation | +| Turn lifecycle | None | `SpeechStarted` → audio → `Flushed` → `SpeechMetadata` per turn (server-assigned `speech_id`) | +| Cross-turn context | None (reconnect to reset) | Prosody persists across turns automatically — no API surface | +| Transports | REST + WebSocket | REST (batch) + WebSocket (streaming) | +| Streaming encodings | `linear16`, `mulaw`, `alaw` | `linear16`, `mulaw`, `alaw` — raw audio only | +| Batch encodings | `mp3`, `opus`, `flac`, `aac`, `linear16`, `mulaw`, `alaw` + `container` / `bit_rate` | Same — but batch-only; the socket rejects them | +| Interruption | `Clear` discards the buffer, no feedback | `Interrupt` → `SpeechInterrupted` with `text_spoken` / `text_remaining` | +| Mid-stream reconfig | No (fixed at connection) | Yes — `Configure` updates `speed` only | +| `speed` | `0.7` to `1.5`, Aura-2, English and Spanish only | `0.5` to `1.5` in `0.05` steps; capped at `1.15` when the text carries a pause marker (`PAUSE_SPEED_CAP_EXCEEDED` above that); see Inline controls for the pronunciation rule | +| `expressivity` | Not supported | `-2`…`2`, default `0` (beta; fixed for the connection) | +| Inline controls | Pronunciation `\{"word":"...","pronounce":""\}` (GA on Aura-2, English and Spanish, input up to 2000 characters, combinable with `speed`); no pause control | Pronunciation (Early Access, both transports) only with `speed` exactly `1.0`: `CONTROL_COMBINATION_INVALID` on batch, `DATA-0002` on the socket; pause `\{pause:500ms\}` on batch only, 500 to 3000 ms in 100 ms steps, at most 8 per request | +| Voice Agent `provider.version` | `v1` (the default when a provider is specified) | `v2` (required) | + +**Pick Aura (`/v1/speak`) when:** +- You need a language other than English, or a specific Aura voice +- You need compressed output (`mp3`, `opus`, `flac`, `aac`) inside a Voice Agent, where Flux TTS returns `INVALID_SETTINGS`; on batch REST both families serve those encodings +- You're already on Aura and nothing in Flux TTS is pulling you over — v1 is unchanged + +**Pick Flux TTS (`/v2/speak`) when:** +- Building a voice agent, phone assistant, or customer-service bot +- You're streaming LLM tokens to a speaker in real time and want the lowest time-to-first-audio +- The user may barge in mid-response and you need to know what they actually heard +- You want tone to carry across turns without managing state yourself +- You're pre-rendering fixed audio (IVR prompts, notifications) with a Flux TTS voice — use the batch REST transport + +Migrating from Aura? See the official [Migrating from Aura to Flux TTS](https://developers.deepgram.com/docs/flux-tts/migrating) guide and [Batch vs Streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming). + +## API Domains + +| Domain | REST | WebSocket | Reference | +|--------|------|-----------|-----------| +| Listen v1 — STT, Nova models | `POST /v1/listen` | `wss://api.deepgram.com/v1/listen` | [listen.md](references/listen.md) | +| Listen v2 — STT, Flux STT (conversational) | — | `wss://api.deepgram.com/v2/listen` | [listen.md](references/listen.md) | +| Speak v1 — TTS, Aura models | `POST /v1/speak` | `wss://api.deepgram.com/v1/speak` | [speak.md](references/speak.md) | +| Speak v2 — TTS, Flux TTS (turn-based) | `POST /v2/speak` | `wss://api.deepgram.com/v2/speak` | [speak.md](references/speak.md) | +| Voice Agent | `GET agent.deepgram.com/v1/agent/settings/think/models`; reusable agent configurations at `/v1/projects/{project_id}/agents` (`GET`, `POST`) and `/v1/projects/{project_id}/agents/{agent_id}` (`GET`, `PUT`, `DELETE`); agent variables at `/v1/projects/{project_id}/agent-variables` (`GET`, `POST`) and `/v1/projects/{project_id}/agent-variables/{variable_id}` (`GET`, `PATCH`, `DELETE`) | `wss://agent.deepgram.com/v1/agent/converse` | [agent.md](references/agent.md) | +| Read (Intelligence) | `POST /v1/read` | — | [read.md](references/read.md) | +| Models | `GET /v1/models`, `GET /v1/models/{model_id}`, `GET /v1/projects/{project_id}/models`, `GET /v1/projects/{project_id}/models/{model_id}`; `include_outdated=true` on either list call also returns non-latest model versions | none | [models.md](references/models.md) | +| Projects | `/v1/projects/*` | — | [projects.md](references/projects.md) | +| Auth | `POST /v1/auth/grant` | — | [auth.md](references/auth.md) | +| Self-Hosted | `/v1/projects/*/self-hosted/*` | — | [self-hosted.md](references/self-hosted.md) | + +## Common Mistakes to Avoid + +### All APIs + +1. **Feature flags are query params, except for Voice Agent and the v2 mid-session updates.** For `/v1/listen`, `/v2/listen`, `/v1/speak`, and `/v2/speak`, initial options go on the URL. For Listen, the request body carries audio (REST) or audio frames (WebSocket); for Speak, it carries text (REST JSON `text` field or WebSocket `Speak` messages). Exceptions: `/v1/agent/converse` has no URL query params at all (all config goes in the `Settings` message); `/v2/listen` supports a `Configure` message after connection to update EOT thresholds, keyterms, language hints, and `numerals` mid-session; and `/v2/speak` supports a `Configure` message that updates `speed` only. Also note that `/v2/listen` has a much smaller param set than `/v1/listen`: flags like `smart_format`, `diarize_model`, and `punctuate` are not available. + +2. **Rate limits are concurrent connections, not total requests.** A 429 means too many simultaneous open connections, not too high a request volume. Diarization and other compute-heavy features reduce your concurrency allowance further. Limits apply per project, not per API key, and differ by region; the [API Rate Limits](https://developers.deepgram.com/reference/api-rate-limits) page carries the per-region concurrency tables. + +### STT WebSocket (`/v1/listen`) + +3. **Send KeepAlive as a text frame, not binary.** The connection closes after 10 seconds of no audio. Send `{"type":"KeepAlive"}` as a text (JSON) frame every 3–5 seconds during silence. Sending it as a binary frame causes transcription delays — the audio pipeline chokes — not a silent no-op. + +4. **Never send empty byte payloads.** Sending a zero-length binary frame to `/v1/listen` is treated as a close — it terminates the connection. Always check that your audio packet has length before sending. + +5. **`encoding` must match the actual audio format.** If `encoding=linear16` but you're sending opus, you'll get a DATA-0000 error or garbled output. Omit `encoding` entirely when sending containerized formats (mp3, wav, ogg) — Deepgram detects them automatically. + +6. **Timestamps reset on reconnect.** Each new WebSocket connection restarts timestamps at 00:00:00. For real-time apps, maintain a timestamp offset across reconnections or you'll silently corrupt your transcript timeline. + +### TTS WebSocket (`/v1/speak`) + +7. **Don't send empty text.** A `Speak` message with an empty `text` field returns a 400 error. Always validate input before sending. + +8. **Character rate limiting (DATA-0001) means slow down, not retry.** If you hit this, reduce how fast you're submitting text chunks — don't immediately retry or you'll compound the problem. + +### Flux TTS (`/v2/speak`) + +9. **`model` is required, and must be a `flux-*` voice.** Unlike `/v1/speak` there is no default — a connection or request without `model` is rejected. Aura strings are rejected on `/v2/speak`, and Flux voices are not served by `/v1/speak`; the two families never mix. Model strings are `flux-{voice}-{language}`, e.g. `flux-alexis-en`. There is no version segment — generations roll forward behind a stable name, as with Flux STT. + +10. **`Flush` ends the turn — it is not a v1-style buffer flush.** There is no `Finalize`; it's folded into `Flush`. Audio starts streaming on its own before you flush, so don't wait to send text. Use the turn's `SpeechMetadata` (not `Flushed`) as your end-of-turn signal — it arrives once all of the turn's audio has been sent, and carries the billing and timing counts, so you can drop client-side character or duration tracking. The server assigns the turn's `speech_id`; never send one yourself. + +11. **Streaming is raw audio only, and rejects anything it doesn't recognize.** The WebSocket emits non-containerized audio, so `encoding` is limited to `linear16` (default), `mulaw`, or `alaw`. The compressed and containerized encodings (`mp3`, `opus`, `flac`, `aac`) and the `container`, `bit_rate`, `callback`, `callback_method`, and `priority` params are **batch-only** — sending them to the socket fails the connection, as does any unknown or misspelled param. Use the batch REST transport when you need compressed output. + +12. **Insert whitespace between separate generations, because the server won't.** Text normalization runs before synthesis, but successive `Speak` messages are concatenated verbatim. Sending `"Hello world."` then `"How are you?"` is processed as `"Hello world.How are you?"`, which causes sentence-boundary artifacts. Add a single space (or the right separator for non-whitespace languages) when you stitch a reply, a tool-call result, and another reply together. Send plain text: SSML is not interpreted, and the only markup Flux TTS honors is its own escaped inline controls. A pronunciation override `\{"word":"...","pronounce":""\}` is honored on both transports (Early Access) but only with `speed` 1.0, and a pause marker `\{pause:500ms\}` is batch-only. A pause marker on the socket, or a pronunciation control on a socket whose `speed` is not 1.0, fails the connection with `DATA-0002`. On batch `POST /v2/speak` the same violations are a 400 whose `err_code` names the rule: `CONTROL_COMBINATION_INVALID` (pronunciation with a pause, or with a `speed` other than `1.0`), `PAUSE_SPEED_CAP_EXCEEDED` (a pause marker with `speed` above `1.15`), `BREAK_OUT_OF_RANGE` (a pause outside 500 to 3000 ms), `BREAK_INCREMENT_INVALID` (a pause off the 100 ms grid), `BREAKS_LIMIT_EXCEEDED` (more than 8 pause markers, or two with no text between them), and `BREAK_SYNTAX_INVALID` (a malformed marker, such as a simple marker without backslashes or an escaped structured marker). A `speed` of exactly `1.0` never counts as a speed control, so it triggers none of these. See [Speed, Pause, Pronunciation](https://developers.deepgram.com/docs/tts-voice-controls). + +### Voice Agent (`/v1/agent/converse`) + +13. **Send the `Settings` message before any audio.** The agent ignores everything until it receives and acknowledges the Settings configuration. Message ordering is strictly required. + +14. **`agent.speak.provider.version` selects the TTS family — and omitting `agent.speak` now gives you Flux TTS.** Set `version` to `v2` for Flux TTS or `v1` for Aura; when you specify a provider but omit `version`, it defaults to `v1`. But if you omit `agent.speak` entirely, the agent defaults to Flux TTS with the `flux-kit-en` voice. Switch families by changing `version` and `model` together — a `flux-*` model under `v1`, or an `aura-*` model under `v2`, is invalid: + ```json + { "agent": { "speak": { "provider": { "type": "deepgram", "version": "v2", "model": "flux-alexis-en" } } } } + ``` + +15. **`GET /v1/agent/settings/think/models` lives on `agent.deepgram.com`, not `api.deepgram.com`.** `GET /v1/agent/settings/think/models`, the list of LLMs you can name in `agent.think.provider`, returns **404 on `api.deepgram.com`** and 200 on `agent.deepgram.com`. Same key, same path; only the host differs, so a client with one hardcoded base URL silently gets a 404 that looks like a missing feature. The three regional `api.*` hosts serve it as well. + +### Flux STT model (`/v2/listen`) + +16. **Use `/v2/listen` and a `flux-general-*` model.** Two are served: `flux-general-en` (English) and `flux-general-multi` (multilingual, and the only model that accepts `language_hint` / `language_hints`). `/v1/listen` does not support Flux STT, and `model=flux` alone is not a valid value. Do not include `language` or `encoding` params for containerized audio. + +17. **Use `Configure` to update EOT thresholds, keyterms, language hints, and `numerals` mid-session.** Unlike `/v1/listen`, Flux STT supports live reconfiguration after connection, so there is no need to reconnect to change turn detection sensitivity, boost new keyterms, re-bias language detection (`language_hints`, `flux-general-multi` only), or switch `numerals` on for a PIN or order number: + ```json + { "type": "Configure", "thresholds": { "eot_threshold": 0.8, "eot_timeout_ms": 3000 }, "keyterms": ["Deepgram"] } + ``` + The server responds with `ConfigureSuccess`, which echoes the full active configuration, `numerals` included, not only the fields you sent, or `ConfigureFailure`, which carries `code` and `description` identifying the rejected configuration. Omitted threshold fields keep their current values. + +18. **`ForceEndTurn` outside a turn is a `Warning`, not an error, and the socket stays open.** Sending `{"type":"ForceEndTurn"}` while no turn is in progress returns `{"type":"Warning","code":"FORCE_END_TURN_NO_ACTIVE_TURN","description":"Received ForceEndTurn while no turn was active; the request was ignored."}` and the connection continues. Do not treat it as fatal or reconnect. `references/listen.md` shows the message shape (`ListenV2Warning`: `code`, `description`, `request_id`, `sequence_id`); `code` is a free string there, so the individual codes such as `FORCE_END_TURN_NO_ACTIVE_TURN` come from the [Force End Turn](https://developers.deepgram.com/docs/flux/force-end-turn) docs. When `ForceEndTurn` *does* land mid-turn, the resulting `TurnInfo` carries `event: "EndOfTurn"` with `trigger: "manual"`. `trigger` is `model` | `manual` | `timeout`, it appears on `EndOfTurn` and nowhere else, and it is an open enum, so tolerate values you do not recognize. + +### Nova diarization (`/v1/listen`) + +19. **Use `diarize_model`, and never send it alongside `diarize`.** `diarize` is deprecated. `diarize_model` both enables diarization and picks the version, so you do not also need `diarize=true` — and sending both fails the request: `400 "diarize_model cannot be used together with diarize or diarize_version."`. Values are `latest`, `v1`, and `v2` for batch (`latest` is currently v2), and `latest` or `v1` for streaming. When diarization is on, `metadata.diarize_info` reports which model actually ran (`{"model_uuid": …, "arch": "v2"}`), which is the only way to tell what `latest` resolved to. + +### Text and Audio Intelligence (`/v1/read`, `/v1/listen`) + +20. **`language` is required on `/v1/read`, and it is validated before anything else.** There is no default: omitting it returns `400 INVALID_QUERY_PARAMETER` with the message "Failed to deserialize query parameters: missing field `language`", which masks every other problem in the request. English only: `language=multi` is rejected, and `en-US` is accepted but echoed back as `en`. Two more `/v1/read` shapes worth knowing: the JSON body takes **exactly one** of `text` or `url` (both or neither gives `PAYLOAD_ERROR`, and `url` must point at a plain-text document, since audio gives `REMOTE_CONTENT_ERROR`), and it is POST-only (`GET` and a WebSocket upgrade both return 405). `summarize` on `/v1/read` accepts `v2` as well as `true`. Result paths differ per endpoint: `/v1/read` returns `results.summary.text`, `/v1/listen` returns `results.summary.short`, so code that handles both has to branch. (`sentiment` maps to `results.sentiments` on both.) + +21. **On the Nova streaming socket, only `detect_entities` works — and the other four fail in three different ways.** `detect_entities=true` is supported and puts `entities` at the **top level** of each `Results` message, beside `channel`, not inside `channel.alternatives[0]`. The other four are prerecorded-only: `summarize` fails the handshake with `400 "Summarization is not available for streaming."`; `topics` and `intents` fail it with `403 UNAUTHORIZED_FEATURES_REQUESTED`, which reads like a key-permissions problem even when the same key's prerecorded `topics`/`intents` calls return 200; and `sentiment` is the trap — the handshake succeeds, no error is ever sent, and sentiment simply never appears in the results. + +### Authentication + +22. **JWT TTL applies only to the initial handshake.** Tokens default to 30 seconds. Once the WebSocket connection is established, the token expiring does not close it — tokens are only needed for the upgrade request. + +## SDK-Specific Skills + +This `api` skill covers the product contracts (endpoints, query params, message shapes) that are identical across SDKs. For **language-idiomatic code** — imports, async patterns, builder APIs, common errors — install the SDK-specific skills. Each Deepgram SDK publishes 7 product skills named `deepgram-{lang}-{product}` (e.g. `deepgram-python-speech-to-text`, `deepgram-js-voice-agent`). The `deepgram-{lang}-` prefix avoids collisions when you install skills from multiple SDKs. + +```bash +# Install all skills from a specific SDK +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET + +# Or install a specific product skill from one SDK (note the deepgram-{lang}- prefix) +npx skills add deepgram/deepgram-python-sdk --skill deepgram-python-speech-to-text +npx skills add deepgram/deepgram-js-sdk --skill deepgram-js-voice-agent +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +| Skill | Purpose | +|---|---| +| `recipes` | Minimal runnable snippets per feature per language | +| `examples` | Full integration examples with third-party platforms (Twilio, LiveKit, etc.) | +| `starters` | Runnable starter apps (framework × feature matrix) | +| `docs` | Navigate Deepgram documentation | +| `audio-intelligence` | The `summarize`, `sentiment`, `topics`, `intents`, and `detect_entities` parameters on `/v1/listen` | +| `text-intelligence` | `POST /v1/read` for text you already have | +| `browser-agent` | The Browser Agent SDK packages for running an agent in a browser | +| `cli` | `deepctl` for shell and CI work | +| `self-hosted` | Running Deepgram on your own GPUs | +| `setup-mcp` | Install the Deepgram MCP server | + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) +- [Speech-to-Text Getting Started](https://developers.deepgram.com/docs/stt/getting-started) +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS Models](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Audio Intelligence](https://developers.deepgram.com/docs/audio-intelligence) +- [Self-Hosted Deployments](https://developers.deepgram.com/docs/self-hosted-introduction) +- [Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints) +- [Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints) +- [API Rate Limits](https://developers.deepgram.com/reference/api-rate-limits): per-region concurrency tables for every API; limits apply per project, not per API key +- [Working with Concurrency Rate Limits](https://developers.deepgram.com/docs/working-with-concurrency-rate-limits) + + +--- + +--- +name: docs +description: > + Find the right Deepgram documentation for any task. Use whenever someone needs help locating + docs, understanding which API to use, or wants to ask questions about Deepgram. Covers all + product areas: speech-to-text (Nova, Flux STT), text-to-speech (Aura, Flux TTS), voice agents, + audio intelligence, and self-hosted deployments. +--- + +# Deepgram Documentation + +Find the right docs for what you're building with Deepgram. + +## Ask AI + +Have a question? Get answers from Deepgram's AI assistant at . + +## Documentation by Topic + +### Speech-to-Text (STT) + +Transcribe audio and video into text. Deepgram ships two actively maintained, next-gen model families — pick the one that matches your use case. + +- **Nova** (`/v1/listen`) — general-purpose transcription (captions, subtitles, batch files, live streams). Rich feature set including intelligence overlays (diarize, summarize, sentiment, topics, intents). +- **Flux STT** (`/v2/listen`) — conversational-audio transcription for voice agents and interactive assistants. Built-in turn-taking (EOT events, mid-session reconfig). + +Docs: +- [STT Getting Started (Nova)](https://developers.deepgram.com/docs/stt/getting-started) +- [Flux STT Quickstart](https://developers.deepgram.com/docs/flux/quickstart) +- [Nova 3 → Flux STT migration](https://developers.deepgram.com/docs/flux/nova-3-migration) +- [Flux STT language prompting](https://developers.deepgram.com/docs/flux/language-prompting) + +### Text-to-Speech (TTS) + +Convert text into natural-sounding speech. Deepgram ships two TTS model families on separate endpoints — the voices do not overlap. + +- **Aura** (`/v1/speak`) — the broadest voice catalog (English, Spanish, German, Dutch, French, Italian, Japanese) and compressed/containerized output. Use for one-shot synthesis and any non-English voice. +- **Flux TTS** (`/v2/speak`) — streaming-first, voice-agent-first synthesis. Turn-based lifecycle, barge-in with spoken-text feedback, and prosody that carries across turns. English at launch. + +Docs: +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Aura voices and languages](https://developers.deepgram.com/docs/tts-models) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Flux TTS Streaming Quickstart](https://developers.deepgram.com/docs/flux-tts/quickstart) +- [Flux TTS Batch (REST) Quickstart](https://developers.deepgram.com/docs/flux-tts/batch) +- [Flux TTS batch vs streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming) +- [Flux TTS voices](https://developers.deepgram.com/docs/flux-tts/voices) +- [Aura → Flux TTS migration](https://developers.deepgram.com/docs/flux-tts/migrating) + +### Voice Agent + +Build conversational voice agents powered by Deepgram. + +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS models (Aura vs Flux TTS)](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Build a Flux TTS voice agent](https://developers.deepgram.com/docs/flux-tts/voice-agent) + +### Text and Audio Intelligence + +Analyze text and audio for sentiment, topics, intents, summaries, and more. + +- [Audio Intelligence Docs](https://developers.deepgram.com/docs/audio-intelligence) + +### Self-Hosted Deployments + +Run Deepgram on your own infrastructure. + +- [Self-Hosted Introduction](https://developers.deepgram.com/docs/self-hosted-introduction) + +### API Reference + +Full reference for all Deepgram REST and WebSocket APIs. + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## SDK-Specific Skills + +For language-idiomatic code patterns (imports, async idioms, error handling, type shapes), install the Deepgram SDK's own skills. Every Deepgram SDK publishes 7 product skills: + +```bash +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +- `api`: consolidated REST + WebSocket API reference +- `recipes`: minimal runnable feature snippets per language +- `examples`: full integration examples with third-party platforms +- `starters`: runnable starter apps (framework × feature) +- `audio-intelligence`: the `/v1/listen` analysis parameters +- `text-intelligence`: `POST /v1/read` for text you already have +- `browser-agent`: running a voice agent in a browser +- `cli`: `deepctl` for shell and CI work +- `self-hosted`: running Deepgram on your own GPUs +- `setup-mcp`: Deepgram MCP server installation + +## MCP Server + +For direct documentation querying from your AI coding tool, use the `setup-mcp` skill to install the Deepgram MCP server. + + +--- + +--- +name: setup-mcp +description: > + Set up a Deepgram MCP server for your AI coding tool. Offers three paths: the Deepgram CLI + MCP proxy (dg mcp), the standalone deepgram-mcp package, and the credential-free hosted + documentation MCP. Use whenever someone wants to install Deepgram's agentic tools, set up + the MCP server, or connect their editor to Deepgram. +--- + +# Install a Deepgram MCP Server + +You are setting up Deepgram MCP integration for the user. Follow these steps. + +## Step 1: Pick a path + +Three paths exist. Pick by whether the user has, or wants, a Deepgram API key. + +| Path | Server | Credentials | Install footprint | +|---|---|---|---| +| **A** | Deepgram CLI MCP proxy (`dg mcp`) | Deepgram API key **required** | Full CLI (`deepctl`) | +| **B** | Standalone `deepgram-mcp` | Deepgram API key **required** | One Python package | +| **C** | Hosted docs MCP (`/_mcp/server`) | **None** | Nothing to install | + +Decision rule: + +- The user already has the CLI, or wants `dg listen` / `dg speak` / `dg init` too → **Path A**. +- The user has an API key but wants only the MCP server, no CLI → **Path B**. +- The user has no API key, or wants something working in one command → **Path C**. + +A key-authenticated hosted variant of Paths A/B also exists at `api.dx.deepgram.com/kapa/mcp`, +with nothing to install — see "The kapa endpoints are not credential-free" below. + +Paths A and B are the same server: `dg mcp` wraps the `deepgram-mcp` package. Both proxy +Deepgram's developer API and fetch their tool list from Deepgram at runtime, so new tools +appear on reconnect without a package upgrade. As of this writing that list is a single +documentation and knowledge-source search tool (`search_deepgram_knowledge_sources`) — check +`tools/list` in the user's client for what is live rather than promising a tool set. + +Paths A/B and Path C both answer Deepgram questions from documentation, so installing more +than one is usually redundant. Path C is the only one that works with no credentials. + +## Step 2: Detect the environment + +Determine which AI coding tool the user is running. Check for: + +- **Claude Code** — look for a `.claude/` directory in the project or user home +- **Cursor** — look for a `.cursor/` directory in the project root +- **Windsurf** — look for a `.windsurf/` directory in the project root + +If multiple are detected, or none are detected, ask the user which tool they want to configure. + +## Step 3: Ask about scope + +Ask the user whether they want the MCP server configured: + +- **For this project only** (recommended for team repos) +- **Globally** (available in all projects) + +--- + +## Path A — Deepgram CLI MCP proxy (`dg mcp`) + +### A1. Install the CLI + +Check first: `dg --version` (or `deepctl --version`, or `where dg` on Windows). The package is +`deepctl` and installs three interchangeable binaries — `dg`, `deepctl`, and `deepgram`. + +```sh +# macOS / Linux — Homebrew (also brings in ffmpeg and portaudio) +brew install deepgram/tap/deepgram + +# macOS / Linux — install script +curl -fsSL https://deepgram.com/install.sh | sh + +# pip / uv / pipx +pip install deepctl +uv tool install deepctl +pipx install deepctl +``` + +```powershell +# Windows — PowerShell +iwr https://deepgram.com/install.ps1 -useb | iex +``` + +To upgrade, use the installer that put it there: `pip install -U deepctl`, +`uv tool upgrade deepctl`, `pipx upgrade deepctl`, `brew upgrade deepgram`, or re-run the install +script. `dg update --check-only` reports whether a newer release exists; on a pip install, bare +`dg update` reports `installation_method: null` instead of upgrading. + +The fully qualified Homebrew name matters: Homebrew 6 loads a third-party formula only after it +is trusted, and `brew install deepgram/tap/deepgram` trusts that one formula, where +`brew tap deepgram/tap && brew install deepgram` fails until a separate `brew trust` step. The +tap formula pins `deepctl-0.2.26`; pip, uv, and pipx install 0.3.1. + +### A2. Authenticate — required + +`dg mcp` will not start without credentials. Do this before configuring any editor: + +```sh +dg login # interactive; or dg login --api-key +dg whoami # confirm: "authenticated": true +``` + +`DEEPGRAM_API_KEY` in the environment works instead of `dg login`. Get a key at +. + +### A3. Configure the editor + +#### Claude Code + +```sh +# Project scope +claude mcp add deepgram --scope project dg mcp + +# User/global scope +claude mcp add deepgram dg mcp +``` + +#### Cursor + +Write or merge into the project's `.cursor/mcp.json`: + +```json +{ + "mcpServers": { + "deepgram": { + "type": "stdio", + "command": "dg", + "args": ["mcp"] + } + } +} +``` + +#### Windsurf + +Write or merge into the project's `.windsurf/mcp.json`, using the same object as Cursor above. + +#### Without a permanent install + +`uvx` and `pipx run` fetch `deepctl` on demand. Credentials still come from `dg login` or +`DEEPGRAM_API_KEY`: + +```json +{ + "mcpServers": { + "deepgram": { + "command": "uvx", + "args": ["deepctl", "mcp"] + } + } +} +``` + +#### Other tools + +- **Transport:** stdio +- **Command:** `dg` +- **Args:** `["mcp"]` + +`dg mcp --transport sse --port 8000` serves SSE instead, for clients that need HTTP. + +--- + +## Path B — Standalone `deepgram-mcp` + +The MCP server without the rest of the CLI. One package, one binary. + +```sh +pip install deepgram-mcp +export DEEPGRAM_API_KEY=your_key_here +``` + +`deepgram-mcp` is a PyPI package. The npm package of the same name is unrelated third-party code +that also asks for `DEEPGRAM_API_KEY`, so do not run `npx deepgram-mcp`. + +#### Claude Code + +```sh +claude mcp add deepgram -- deepgram-mcp +``` + +#### Cursor / Windsurf + +Write or merge into `.cursor/mcp.json` or the Windsurf MCP config: + +```json +{ + "mcpServers": { + "deepgram": { + "command": "deepgram-mcp", + "env": { + "DEEPGRAM_API_KEY": "your_key_here" + } + } + } +} +``` + +`--api-key` overrides the environment variable, and `--transport sse --port 8000` serves SSE. +Source: . + +--- + +## Path C — Hosted documentation MCP (no credentials) + +Use `https://developers.deepgram.com/_mcp/server`. It answers unauthenticated, needs no API +key, and exposes one tool, `searchDocs`, which returns documentation passages with source URLs. + +It is not a plain liveness URL. `HEAD` returns 404, a `GET` with the MCP +`Accept: application/json, text/event-stream` header returns 405, and a bare `GET` returns a +JSON descriptor of the server rather than an MCP response. Only a `POST` `initialize` exercises +the server; it answers 200 with `serverInfo.name` `fern-docs-mcp-server`: + +```sh +curl -s -X POST https://developers.deepgram.com/_mcp/server \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-03-26","capabilities":{},"clientInfo":{"name":"probe","version":"0"}}}' +``` + +#### Claude Code + +```sh +# Project scope +claude mcp add deepgram-docs --scope project --transport http https://developers.deepgram.com/_mcp/server + +# User/global scope +claude mcp add deepgram-docs --transport http https://developers.deepgram.com/_mcp/server +``` + +#### Cursor + +Write or merge into the project's `.cursor/mcp.json`: + +```json +{ + "mcpServers": { + "deepgram-docs": { + "type": "http", + "url": "https://developers.deepgram.com/_mcp/server" + } + } +} +``` + +#### Windsurf + +Write or merge into the project's `.windsurf/mcp.json`, using the same object as Cursor above. + +#### Other tools + +- **Type:** HTTP +- **URL:** `https://developers.deepgram.com/_mcp/server` + +### The kapa endpoints are not credential-free + +`https://api.dx.deepgram.com/kapa/mcp` and `https://deepgram.mcp.kapa.ai` both exist, and both +reject an unauthenticated request with HTTP 401 plus a `WWW-Authenticate: Bearer +resource_metadata=...` header, so a client that implements MCP's OAuth flow can connect to either. +They differ in whether a Deepgram API key works: + +- **`api.dx.deepgram.com/kapa/mcp` accepts a Deepgram API key.** Send it as either + `Authorization: Token ` or `Authorization: Bearer ` and `initialize` returns 200 from + `deepgram-mcp-relay`; an invalid key gets 401. `tools/list` returns the same single + `search_deepgram_knowledge_sources` tool as Paths A and B, so this is the hosted HTTP form of + the same server — useful when the user has a key but cannot install anything. Pass the key as a + header, or the client falls back to OAuth: + + ```sh + claude mcp add deepgram-relay --transport http https://api.dx.deepgram.com/kapa/mcp \ + --header "Authorization: Token $DEEPGRAM_API_KEY" + ``` +- **`deepgram.mcp.kapa.ai` does not.** A Deepgram API key gets 401 with either scheme. OAuth is + the only way in. + +Neither is the zero-setup option — use `/_mcp/server` for that. + +--- + +## Step 4: Confirm + +- **Claude Code** — run `/reload-plugins` to activate immediately, no restart needed. +- **Cursor / Windsurf / Other** — the user may need to restart or reload their tool. + +Then tell the user the server is configured, and check what it actually exposes before +describing it — have the client list its tools rather than naming tools from memory. + +For Path C, add: + +> Your tool can now search Deepgram's documentation directly — try asking about API +> parameters, voice agents, or model capabilities. + +Link them to [Deepgram Agentic Tools](https://developers.deepgram.com/developer-tools/agentic-tools) +for more details. Its two kapa URLs, `https://api.dx.deepgram.com/kapa/mcp` and +`https://deepgram.mcp.kapa.ai`, require credentials: an unauthenticated `initialize` returns 401. +The Docs MCP server at `https://developers.deepgram.com/_mcp/server` is the credential-free path. + +## Troubleshooting + +**`Error: DEEPGRAM_API_KEY is not set in the configuration file (...config.yaml) or environment variable.`** +followed by `Run deepctl login to configure the CLI with your Deepgram account.` +→ Path A with no credentials. `dg mcp` exits 1 before serving anything. Run `dg login`, or set +`DEEPGRAM_API_KEY`. Confirm with `dg whoami`. + +**`Error: No API key. Set DEEPGRAM_API_KEY or use --api-key.`** +→ Path B with no credentials. Export `DEEPGRAM_API_KEY`, put it in the server's `env` block, or +pass `--api-key`. + +**`! Needs authentication` in `claude mcp list`, or HTTP 401 `{"status_code":401,"detail":"Authentication required"}` / `{"error":"invalid_token"}`** +→ You are pointed at a kapa endpoint with no credentials. Switch to +`https://developers.deepgram.com/_mcp/server`, which needs none. To stay on +`api.dx.deepgram.com/kapa/mcp`, add `--header "Authorization: Token $DEEPGRAM_API_KEY"` — that +endpoint accepts a Deepgram API key. On `deepgram.mcp.kapa.ai` an API key does not work; let the +client run its OAuth flow instead. + +**`Server "deepgram-docs" is defined in multiple scopes with different endpoints`** +→ An earlier setup registered `deepgram-docs` at a kapa URL in user scope, and this one added a +different URL in project scope. OAuth tokens are stored per endpoint, so authenticating one does +not carry over. Keep one: `claude mcp remove deepgram-docs -s user` (or `-s project`). Check for +a pre-existing entry with `claude mcp get deepgram-docs` before adding, and pick a distinct +server name if the user wants to keep both. + +**`ImportError` mentioning `streamablehttp_client` on startup** +→ An incompatible `mcp` package. `deepgram-mcp` imports `streamablehttp_client` from +`mcp.client.streamable_http`, which `mcp` 2.0 removed. Install into a clean environment, or pin +`mcp>=1.0.0,<2.0.0`. Installing `deepctl` pins this for you. + +**The server connects but exposes fewer tools than expected** +→ Expected. Paths A and B fetch their tool list from Deepgram at runtime, so it reflects what +the API serves right now, not what the package version implies. Reconnect to pick up new tools. + +**Anything else on Path A** +→ Verify `dg --version` works and `dg mcp` runs in a terminal without errors, then +`dg update --check-only` to see whether a newer release exists. + +## Sources + +- Deepgram CLI: +- `deepgram-mcp`: +- Deepgram Agentic Tools: + + +--- + +--- +name: starters +description: > + Clone a ready-to-run Deepgram demo app and start building on top of it. Use whenever someone + wants a quick working demo, needs to prototype with Deepgram, or is starting a new project + that uses speech-to-text, text-to-speech, voice agents, audio intelligence, or live streaming. + Match the user's language, framework, and desired Deepgram feature to the right starter. +--- + +# Deepgram Starter Apps + +Clone a working demo and start building. Every starter is a minimal, runnable app you can extend. + +## 1. Pick Your Feature + +What do you want to build? + +- **Transcribe a file** → `transcription` — send audio/video, get text back (REST, Nova) +- **Transcribe a live stream** → `live-transcription` — real-time speech-to-text (WebSocket, Nova) +- **Generate speech** → `text-to-speech` — send text, get audio back (REST, Aura) +- **Stream speech** → `live-text-to-speech` — real-time text-to-audio (WebSocket, Aura) +- **Analyze text** → `text-intelligence` — sentiment, topics, intents, summaries over text you + already have (REST, `/v1/read`) +- **Build a voice agent** → `voice-agent` — conversational AI agent (WebSocket, agent.deepgram.com) +- **Conversational STT with turn detection** → `flux` — Deepgram Flux STT for voice agents and interactive assistants (WebSocket, `/v2/listen`) +- **Turn-based TTS for a voice agent** → `flux-tts` — Deepgram Flux TTS, streaming synthesis with barge-in (WebSocket, `/v2/speak`) + +**There is no audio-intelligence starter.** `text-intelligence` is text-only — it posts text you +already have to `/v1/read`. No `{framework}-audio-intelligence` repository exists in +`deepgram-starters` for any framework, so don't construct those URLs. To run intelligence features +(summarization, sentiment, topics, intents) over *audio*, they are query parameters on +`/v1/listen`, not a separate starter: clone the `transcription` starter for your framework and add +the parameters to its existing request. See the `api` skill for which features `/v1/listen` +supports. + +**Nova vs Flux STT for speech-to-text:** use `transcription` or `live-transcription` (Nova, `/v1/listen`) for general-purpose transcription, captions, and batch workloads. Use `flux` (Flux STT, `/v2/listen`) when you need built-in turn detection for conversational audio. See the `api` skill for a full comparison. + +**Aura vs Flux TTS for text-to-speech:** use `text-to-speech` or `live-text-to-speech` (Aura, `/v1/speak`) for one-shot synthesis, non-English voices, and compressed audio. Use `flux-tts` (Flux TTS, `/v2/speak`) when you're streaming LLM output to a speaker and need a turn lifecycle and barge-in. See the `api` skill for a full comparison. + +**Flux TTS starters exist for `node`, `flask`, `fastapi`, `django`, and `java` only** — these are the five apps Deepgram officially publishes at [Flux TTS template apps](https://developers.deepgram.com/docs/flux-tts/template-apps). There is no `flux-tts` starter for the other frameworks; don't construct those URLs. For an unsupported framework, start from the `api` skill's Flux TTS section and the SDK skills instead. + +## 2. Pick Your Stack + +| Language | Frameworks | +|----------|------------| +| JavaScript | `node` | +| TypeScript | `bun`, `deno` | +| Python | `fastapi`, `flask`, `django` | +| Go | `go` | +| Java | `java` | +| C# | `csharp` | +| Rust | `rust` | +| Ruby | `ruby` | +| PHP | `php` | +| C++ | `cpp` | + +## 3. Clone and Run + +Every starter lives at `https://github.com/deepgram-starters/{framework}-{feature}` — framework +first, feature second. Clone **with submodules**; each starter vendors two git submodules — its +browser frontend at `frontend/` and the shared starter contracts at `contracts/` — and a plain +`git clone` leaves both directories empty and the app unrunnable: + +```sh +git clone --recurse-submodules https://github.com/deepgram-starters/{framework}-{feature}.git +cd {framework}-{feature} +``` + +In 80 of the 96 starters, both submodule URLs in `.gitmodules` are SSH (`git@github.com:...`) +even though both repositories are public, so `--recurse-submodules` fails with +`Host key verification failed` unless the user has a GitHub SSH key. The other 16 use HTTPS URLs +and clone without a key: 12 of the 13 `{framework}-live-transcription` starters (every one except +`rust-live-transcription`) plus `csharp-voice-agent`, `django-voice-agent`, `flask-voice-agent`, +and `node-voice-agent`. Without an SSH key, rewrite SSH to HTTPS for the clone. The rewrite +changes nothing on the 16 HTTPS starters, so it is safe to use on every starter: + +```sh +git -c url."https://github.com/".insteadOf="git@github.com:" \ + clone --recurse-submodules https://github.com/deepgram-starters/{framework}-{feature}.git +``` + +The starter's own `make init` runs `git submodule update --init --recursive` and installs +dependencies, but it inherits the URLs in `.gitmodules`. On the 80 SSH starters it fails +identically without a key, so it is the path for users who **have** SSH set up (or for one of the +16 HTTPS starters), not a workaround for users who don't. + +Set your API key and follow the README: + +```sh +export DEEPGRAM_API_KEY=your_key_here +``` + +Get an API key at . + +### Or scaffold with the CLI + +The [Deepgram CLI](https://github.com/deepgram/cli) has a scaffolder that finds and clones a +starter for you: + +```sh +dg init --list # browse templates +dg init --list --search python # filter +dg init node-transcription # clone into ./node-transcription +dg init node-transcription --dir ./my-app +``` + +**`dg init` does not solve the submodule problem.** It runs a plain clone, so `frontend/` and +`contracts/` land empty, and it still prints `Done! … is ready` and `"status": "success"`. Adding +`--install` runs the starter's `make check-prereqs && make init`, which hits the same `.gitmodules` +URLs: on the 80 SSH starters it fails with `Host key verification failed`, and `dg init` reports +success anyway. Without a GitHub SSH key, finish the checkout by hand after `dg init`: + +```sh +cd my-app +git -c url."https://github.com/".insteadOf="git@github.com:" \ + submodule update --init --recursive +``` + +`dg init` is also marked alpha, and its templates gallery is a separate list from the matrix +below rather than a subset of it. It carries 44 templates with no `flux` or `flux-tts` entries; +it still lists `sinatra-transcription`, whose repository is archived and private, so the clone +returns 404 for anyone outside Deepgram; and it lists `nextjs-*` templates that now redirect out +of `deepgram-starters` to `deepgram-devs`, which is why there is no `nextjs` row below. Treat +the matrix as authoritative and fall back to `git clone`. See the `cli` skill for installing +`deepctl` and for the rest of `dg init`. + +## The `{feature}-html` repos are not starters + +The `deepgram-starters` org also contains `transcription-html`, `live-transcription-html`, +`text-to-speech-html`, `live-text-to-speech-html`, `text-intelligence-html`, `voice-agent-html`, +`flux-html`, and `flux-tts-html`. **Do not clone these and do not offer them as starters.** Each +is the shared browser frontend that a backend starter pulls in as its `frontend/` submodule — +`node-transcription` vendors `transcription-html`, `flask-voice-agent` vendors `voice-agent-html`, +`node-flux-tts` and `java-flux-tts` both vendor `flux-tts-html`, and so on. Seven of the eight +say so in their own README ("This is a frontend submodule - do not use directly"); `flux-tts-html` +carries no such warning but is vendored the same way. None of them serve an API, so none of them +run standalone. Clone the backend starter instead and the right frontend arrives with it. + +They also invert the naming rule. The starter pattern is `{framework}-{feature}`, but these are +`{feature}-html` — and the mirror-image names do **not** exist, so do not construct them: +`deepgram-starters/html-transcription` is a 404. There is no vanilla-HTML row in the matrix +because there is no standalone browser starter; for browser-only work, clone the `node` starter +for the feature you want and read its `frontend/` directory. + +## Examples + +**"I want to build a voice agent in Python"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/fastapi-voice-agent.git` + +**"I need live transcription in my Node app"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/node-live-transcription.git` + +**"I want to add text-to-speech to my Go service"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/go-text-to-speech.git` + +**"I want to analyze audio for sentiment in C#"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/csharp-text-intelligence.git` + +**"I want streaming TTS with barge-in for my Node voice agent"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/node-flux-tts.git` + +**"I want a plain browser/HTML demo"** +→ There is no standalone HTML starter. Clone `node-{feature}` and work in its `frontend/` +directory — that is the same browser code the `{feature}-html` submodule holds. + +## All Starters + +Every URL below is a real, published, non-archived repository, and the table is the complete +set: 13 frameworks × 7 features, plus `flux-tts` for the five frameworks that have it. A cell +showing `—` means that starter does not exist; don't construct the URL. + +The `java-flux-tts` README clones with a plain `git clone`, without `--recurse-submodules`, while +its `.gitmodules` points both submodules at SSH URLs, so following its Maven steps leaves +`frontend/` and `contracts/` empty. Use the clone command in section 3 instead. + +| | transcription | live-transcription | text-to-speech | live-text-to-speech | text-intelligence | voice-agent | flux | flux-tts | +|---|---|---|---|---|---|---|---|---| +| **node** | [repo](https://github.com/deepgram-starters/node-transcription) | [repo](https://github.com/deepgram-starters/node-live-transcription) | [repo](https://github.com/deepgram-starters/node-text-to-speech) | [repo](https://github.com/deepgram-starters/node-live-text-to-speech) | [repo](https://github.com/deepgram-starters/node-text-intelligence) | [repo](https://github.com/deepgram-starters/node-voice-agent) | [repo](https://github.com/deepgram-starters/node-flux) | [repo](https://github.com/deepgram-starters/node-flux-tts) | +| **bun** | [repo](https://github.com/deepgram-starters/bun-transcription) | [repo](https://github.com/deepgram-starters/bun-live-transcription) | [repo](https://github.com/deepgram-starters/bun-text-to-speech) | [repo](https://github.com/deepgram-starters/bun-live-text-to-speech) | [repo](https://github.com/deepgram-starters/bun-text-intelligence) | [repo](https://github.com/deepgram-starters/bun-voice-agent) | [repo](https://github.com/deepgram-starters/bun-flux) | — | +| **deno** | [repo](https://github.com/deepgram-starters/deno-transcription) | [repo](https://github.com/deepgram-starters/deno-live-transcription) | [repo](https://github.com/deepgram-starters/deno-text-to-speech) | [repo](https://github.com/deepgram-starters/deno-live-text-to-speech) | [repo](https://github.com/deepgram-starters/deno-text-intelligence) | [repo](https://github.com/deepgram-starters/deno-voice-agent) | [repo](https://github.com/deepgram-starters/deno-flux) | — | +| **fastapi** | [repo](https://github.com/deepgram-starters/fastapi-transcription) | [repo](https://github.com/deepgram-starters/fastapi-live-transcription) | [repo](https://github.com/deepgram-starters/fastapi-text-to-speech) | [repo](https://github.com/deepgram-starters/fastapi-live-text-to-speech) | [repo](https://github.com/deepgram-starters/fastapi-text-intelligence) | [repo](https://github.com/deepgram-starters/fastapi-voice-agent) | [repo](https://github.com/deepgram-starters/fastapi-flux) | [repo](https://github.com/deepgram-starters/fastapi-flux-tts) | +| **flask** | [repo](https://github.com/deepgram-starters/flask-transcription) | [repo](https://github.com/deepgram-starters/flask-live-transcription) | [repo](https://github.com/deepgram-starters/flask-text-to-speech) | [repo](https://github.com/deepgram-starters/flask-live-text-to-speech) | [repo](https://github.com/deepgram-starters/flask-text-intelligence) | [repo](https://github.com/deepgram-starters/flask-voice-agent) | [repo](https://github.com/deepgram-starters/flask-flux) | [repo](https://github.com/deepgram-starters/flask-flux-tts) | +| **django** | [repo](https://github.com/deepgram-starters/django-transcription) | [repo](https://github.com/deepgram-starters/django-live-transcription) | [repo](https://github.com/deepgram-starters/django-text-to-speech) | [repo](https://github.com/deepgram-starters/django-live-text-to-speech) | [repo](https://github.com/deepgram-starters/django-text-intelligence) | [repo](https://github.com/deepgram-starters/django-voice-agent) | [repo](https://github.com/deepgram-starters/django-flux) | [repo](https://github.com/deepgram-starters/django-flux-tts) | +| **go** | [repo](https://github.com/deepgram-starters/go-transcription) | [repo](https://github.com/deepgram-starters/go-live-transcription) | [repo](https://github.com/deepgram-starters/go-text-to-speech) | [repo](https://github.com/deepgram-starters/go-live-text-to-speech) | [repo](https://github.com/deepgram-starters/go-text-intelligence) | [repo](https://github.com/deepgram-starters/go-voice-agent) | [repo](https://github.com/deepgram-starters/go-flux) | — | +| **java** | [repo](https://github.com/deepgram-starters/java-transcription) | [repo](https://github.com/deepgram-starters/java-live-transcription) | [repo](https://github.com/deepgram-starters/java-text-to-speech) | [repo](https://github.com/deepgram-starters/java-live-text-to-speech) | [repo](https://github.com/deepgram-starters/java-text-intelligence) | [repo](https://github.com/deepgram-starters/java-voice-agent) | [repo](https://github.com/deepgram-starters/java-flux) | [repo](https://github.com/deepgram-starters/java-flux-tts) | +| **csharp** | [repo](https://github.com/deepgram-starters/csharp-transcription) | [repo](https://github.com/deepgram-starters/csharp-live-transcription) | [repo](https://github.com/deepgram-starters/csharp-text-to-speech) | [repo](https://github.com/deepgram-starters/csharp-live-text-to-speech) | [repo](https://github.com/deepgram-starters/csharp-text-intelligence) | [repo](https://github.com/deepgram-starters/csharp-voice-agent) | [repo](https://github.com/deepgram-starters/csharp-flux) | — | +| **rust** | [repo](https://github.com/deepgram-starters/rust-transcription) | [repo](https://github.com/deepgram-starters/rust-live-transcription) | [repo](https://github.com/deepgram-starters/rust-text-to-speech) | [repo](https://github.com/deepgram-starters/rust-live-text-to-speech) | [repo](https://github.com/deepgram-starters/rust-text-intelligence) | [repo](https://github.com/deepgram-starters/rust-voice-agent) | [repo](https://github.com/deepgram-starters/rust-flux) | — | +| **ruby** | [repo](https://github.com/deepgram-starters/ruby-transcription) | [repo](https://github.com/deepgram-starters/ruby-live-transcription) | [repo](https://github.com/deepgram-starters/ruby-text-to-speech) | [repo](https://github.com/deepgram-starters/ruby-live-text-to-speech) | [repo](https://github.com/deepgram-starters/ruby-text-intelligence) | [repo](https://github.com/deepgram-starters/ruby-voice-agent) | [repo](https://github.com/deepgram-starters/ruby-flux) | — | +| **php** | [repo](https://github.com/deepgram-starters/php-transcription) | [repo](https://github.com/deepgram-starters/php-live-transcription) | [repo](https://github.com/deepgram-starters/php-text-to-speech) | [repo](https://github.com/deepgram-starters/php-live-text-to-speech) | [repo](https://github.com/deepgram-starters/php-text-intelligence) | [repo](https://github.com/deepgram-starters/php-voice-agent) | [repo](https://github.com/deepgram-starters/php-flux) | — | +| **cpp** | [repo](https://github.com/deepgram-starters/cpp-transcription) | [repo](https://github.com/deepgram-starters/cpp-live-transcription) | [repo](https://github.com/deepgram-starters/cpp-text-to-speech) | [repo](https://github.com/deepgram-starters/cpp-live-text-to-speech) | [repo](https://github.com/deepgram-starters/cpp-text-intelligence) | [repo](https://github.com/deepgram-starters/cpp-voice-agent) | [repo](https://github.com/deepgram-starters/cpp-flux) | — | + +## Need something more specific? + +- **Focused feature snippets** (one feature, one language, < 50 lines) → `recipes` skill → +- **Third-party integrations** (Twilio, LiveKit, LangChain, Vercel AI SDK, Discord, etc.) → `examples` skill → +- **SDK-specific code skills** (idiomatic imports, async patterns, gotchas) → `npx skills add deepgram/deepgram-{lang}-sdk` — see the `api` skill for the 6 SDKs whose skills are publicly installable. + +## Related Deepgram skills + +- `api`: consolidated REST + WebSocket API reference +- `recipes`: minimal runnable feature snippets per language +- `examples`: full integration examples with third-party platforms +- `docs`: documentation finder +- `cli`: `deepctl`, including `dg init` for scaffolding a template from the terminal +- `setup-mcp`: Deepgram MCP server installation + + diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/allowlist.tsv b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/allowlist.tsv new file mode 100644 index 00000000..347443ee --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/allowlist.tsv @@ -0,0 +1,31 @@ +# deepgram/skills main at 0fc13fad726f; writer releases 0.2.16-0.3.2 from PyPI; live_releases: releases that were current on PyPI while main served it (any 0.2.16 to 0.3.2 install could fetch it) +skill bytes sha256 first_commit pushed_at replaced_at live_releases +api 2271 0e84ca7cdbfecde6ccbad869ac1368ffc70e500d7fd63cba317ef5c5a010bc39 d11390a6eefe 2026-03-20T12:09:57Z 2026-04-02T09:38:27Z 0.2.16,0.2.17 +api 6892 1e3c33188e3b6548adac918916e489cecc9dd408cc63eccc7645846a9bf8b5ef 5f0aad96c289 2026-04-02T09:38:27Z 2026-04-02T09:42:39Z 0.2.17 +api 7229 37b0c83a184100354b58dbd6f63fe086e11018aec0e29730a58a71cb72ba8569 b1b11dca4254 2026-04-02T09:42:39Z 2026-04-02T09:50:49Z 0.2.17 +api 7810 87682eb16a5fe904bad30ee1f68c43dc1a6db31217c252e1d2b94e89c8c810ac 1b0682cb517a 2026-04-02T09:50:49Z 2026-04-02T09:58:31Z 0.2.17 +api 7558 ab6dcec901dbe89994ee8b5f43649d488fca95d3ad591d910f6109b5946fea97 977abfe16c99 2026-04-02T09:58:31Z 2026-04-02T09:59:57Z 0.2.17 +api 7769 644f06c0a29a2251a556c5669d26f636074ebfb1ab57f61d7c298d34f5810553 4d1c1bb1bd9c 2026-04-02T09:59:57Z 2026-04-24T12:42:54Z 0.2.17,0.2.18 +api 11915 db3f40de8edb8b810ec9636cf2d5ac8916a4c760555373d5b4f4a75607b9ec59 341586709faf 2026-04-24T12:42:54Z 2026-04-24T14:58:02Z 0.2.18 +api 12087 523e206af4c33a07175d7fd6d190b70ed7b1c7ec89cb3b6b4575669abf02e5a2 316940a6b682 2026-04-24T14:58:02Z 2026-08-13T13:34:13Z 0.2.18,0.2.19,0.2.20,0.2.21,0.2.22,0.2.23,0.2.24,0.2.25,0.2.26 +api 19480 b2855ce6bcc9d8e6744c9b669c8ad0de624100f779a80d9139b73333bfd5e408 9f223f5b21d6 2026-08-13T13:34:13Z 2026-09-18T11:36:59Z 0.2.26,0.3.0 +api 19496 b193fe2baed574077026cf2b60f5d07985ad27899e2e8ffdab6a2657d70601a6 3c5b9904018f 2026-09-18T11:36:59Z 2026-09-18T12:39:39Z 0.3.0 +api 21751 8cda50a65b00eb45b3789fd3e29bc998f7737aa0ad03bb563e782cc3924b556b 84ace660e919 2026-09-18T12:39:39Z 2026-09-18T12:40:42Z 0.3.0 +api 25667 545d78a1b2735479237fe7703128ca033279126772eca2a6dbf1a5a878ea12f2 33b1ce787ef8 2026-09-18T12:40:42Z 2026-09-18T13:06:44Z 0.3.0 +api 26114 f24514384d9662214b973923117802bf5b7329be17b787be0ed5495cb668eab1 62fddda06ca4 2026-09-18T13:06:44Z 2026-10-02T15:52:48Z 0.3.0,0.3.1 +api 29407 959031e436ae0eb50bb5139a30acca1607c3f4828061e655b6b7e3d10be1ea92 0fc13fad726f 2026-10-02T15:52:48Z 0.3.1,0.3.2 +docs 1683 a3d3c853b73b6e0f56cba86a6be91bcba936135d4a58fa0ecc62958de1349e12 d11390a6eefe 2026-03-20T12:09:57Z 2026-04-24T12:42:54Z 0.2.16,0.2.17,0.2.18 +docs 3511 e987e38d0e832c11949a21395c38cec2a0e7c275acfce26e65a0c7c36cffaa0b 341586709faf 2026-04-24T12:42:54Z 2026-08-13T13:34:13Z 0.2.18,0.2.19,0.2.20,0.2.21,0.2.22,0.2.23,0.2.24,0.2.25,0.2.26 +docs 4875 cef47147b79e903c72b3d27a9bd8dcca3ddeb5e0f9f6bbc36da4dd6f13ad8336 9f223f5b21d6 2026-08-13T13:34:13Z 2026-09-18T11:36:59Z 0.2.26,0.3.0 +docs 4925 1c0457b580de0620b8953bb6029873d614eb96e586b7e6be72c83fd63c2e51a8 3c5b9904018f 2026-09-18T11:36:59Z 2026-09-18T13:06:44Z 0.3.0 +docs 5249 64e23bb3edd797bc149a3fd26034170e1c8e2850d608ad15e59915bfa84f289e 62fddda06ca4 2026-09-18T13:06:44Z 0.3.0,0.3.1,0.3.2 +setup-mcp 4647 8ce952a6d4322ea883028c9a548be58fc7178bbba4148baae257aa57d3a7d68a 25bb74c54b4c 2026-03-31T12:58:40Z 2026-09-18T12:39:39Z 0.2.16,0.2.17,0.2.18,0.2.19,0.2.20,0.2.21,0.2.22,0.2.23,0.2.24,0.2.25,0.2.26,0.3.0 +setup-mcp 10674 251db18b9a887660093e82a09b4c3ff02d464e02df1450484a87261d2ce836b3 84ace660e919 2026-09-18T12:39:39Z 2026-10-02T15:52:48Z 0.3.0,0.3.1 +setup-mcp 12301 f5000298802362356907889bfcd90cba430c528d466c3f27bddd3923274d9a54 0fc13fad726f 2026-10-02T15:52:48Z 0.3.1,0.3.2 +starters 8790 9ba321bc8cf444c8b493d290d61e5dda00fb21bb7a56b55bef7b92952a841b2c d11390a6eefe 2026-03-20T12:09:57Z 2026-04-24T12:42:54Z 0.2.16,0.2.17,0.2.18 +starters 10027 6a5622355f3c185b2eafd3dad5b54aca01bda47d11d03dff79f6ab4228194c70 341586709faf 2026-04-24T12:42:54Z 2026-08-13T13:34:13Z 0.2.18,0.2.19,0.2.20,0.2.21,0.2.22,0.2.23,0.2.24,0.2.25,0.2.26 +starters 11464 0074b2b8f7677624ee0a7f94d373085ea70c87e5057b601adf549094e9f61258 9f223f5b21d6 2026-08-13T13:34:13Z 2026-09-18T11:36:59Z 0.2.26,0.3.0 +starters 11489 b92ba4683fc740b77858a3f7b2f9845b6ce07fee91d17ba7ec67d1417b4c70cb 3c5b9904018f 2026-09-18T11:36:59Z 2026-09-18T12:39:39Z 0.3.0 +starters 16472 40c3ec366c632145a619276fee54f426a124a496ebcfcee4b1fa0f34d7a25b9b 84ace660e919 2026-09-18T12:39:39Z 2026-09-18T13:06:44Z 0.3.0 +starters 16572 950c42c6c7c01f5692a5a2cdc00c6e1bbded51dab0e261e290f890f8eda035a0 62fddda06ca4 2026-09-18T13:06:44Z 2026-10-02T15:52:48Z 0.3.0,0.3.1 +starters 17376 aba86630c4872031d3c66dc100e58b3878a3c9b4cbbfad4af87a12102ba8e228 0fc13fad726f 2026-10-02T15:52:48Z 0.3.1,0.3.2 diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/api.md b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/api.md new file mode 100644 index 00000000..9462b7c0 --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/api.md @@ -0,0 +1,332 @@ +--- +name: api +description: > + Deepgram API reference for speech-to-text, text-to-speech, voice agents, audio intelligence, + and account management. Use whenever building with Deepgram APIs — REST or WebSocket. Covers + authentication, all endpoints, query parameters, request/response schemas, and WebSocket + message formats. Reference files are organized by domain: listen (STT — Nova and Flux STT), speak + (TTS — Aura and Flux TTS), agent (voice agents), read (text/audio intelligence), models, + projects, auth, and self-hosted. +--- + +# Deepgram API + +Build with Deepgram's speech-to-text, text-to-speech, voice agent, and audio intelligence APIs. + +> **"Flux" names two separate products.** **Flux STT** is conversational speech-to-text on `/v2/listen` (`model=flux-general-en`). **Flux TTS** is turn-based speech synthesis on `/v2/speak` (`model=flux-{voice}-{language}`). They share a name and a design philosophy — turn-aware, built for voice agents — but they are different endpoints with different models, params, and messages. When a request just says "Flux", check whether it is about transcribing audio or producing it. + +## Getting Started + +All API requests require authentication via API key or JWT: + +- **API Key**: `Authorization: Token ` +- **JWT**: `Authorization: Bearer ` + +Base servers: + +- REST & STT/TTS WebSocket: `https://api.deepgram.com` +- Voice Agent WebSocket **and `GET /v1/agent/settings/think/models`**: `https://agent.deepgram.com` + +`GET /v1/agent/settings/think/models` lives on the `agent.` host too, not on `api.`: it +returns 404 on `api.deepgram.com` and 200 on `agent.deepgram.com`. Everything else REST +stays on `api.deepgram.com`. + +### Regional endpoints + +To keep processing inside a geography, swap the host. Same API keys, same paths, same SDKs — +only the base URL changes. Requests are never routed out of region: if the region is +unavailable they fail rather than fall back. + +| Region | Host | +|---|---| +| EU | `api.eu.deepgram.com` | +| Australia | `api.au.deepgram.com` | +| India | `api.in.deepgram.com` | + +**The data plane is regional; the Projects management API is not.** On all three regional hosts: + +| Endpoint | Regional | +|---|---| +| `POST /v1/listen`, `wss://…/v1/listen` | Yes | +| `wss://…/v2/listen` | Yes | +| `POST /v1/speak`, `wss://…/v1/speak` | Yes | +| `POST /v2/speak`, `wss://…/v2/speak` | Yes | +| `POST /v1/read` | Yes | +| `wss://…/v1/agent/converse` | Yes | +| `GET /v1/models` | Yes | +| `POST /v1/auth/grant` | Yes | +| `/v1/projects/*` (keys, members, usage, billing) | **No — 404** | + +Two host rules that catch people out: + +1. **Voice Agent moves onto the `api.` host regionally.** There is no `agent.eu.deepgram.com` + (the name does not resolve). Use `wss://api.eu.deepgram.com/v1/agent/converse`. `GET /v1/agent/settings/think/models` moves with it. Globally it stays on `agent.deepgram.com`. +2. **Keep management calls on `api.deepgram.com`.** Point a client's management calls at a + regional host and `/v1/projects` returns 404, so split the base URL by call type if your + app both transcribes and manages keys. + +Whisper models are not served in any of the three regions — use Nova or Flux STT models there. + +For Deepgram Dedicated and self-hosted hosts, see +[Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints); +for the full per-region feature matrix and SDK snippets, see +[Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints). + +## How Deepgram's APIs Fit Together + +``` + ┌──────────────────────────────┐ + │ api.deepgram.com │ + └──────────────────────────────┘ + │ + ┌───────────┬───────────┬─────┴─────┬───────────┬───────────┐ + ▼ ▼ ▼ ▼ ▼ ▼ + /v1/listen /v2/listen /v1/speak /v2/speak /v1/read /v1/projects/* + Nova — STT Flux — STT Aura — TTS Flux — TTS Text AI Management + REST + WSS WSS only REST + WSS REST + WSS REST only REST only + + ┌──────────────────────────────┐ + │ agent.deepgram.com │ + └──────────────────────────────┘ + │ + ▼ + /v1/agent/converse + WebSocket only + audio ──▶ STT ──▶ LLM ──▶ TTS ──▶ audio + (Deepgram orchestrates the full pipeline) +``` + +## Which API Should I Use? + +``` +Audio → text (transcription)? +├─ General-purpose transcription (captions, batch, call logs, live streams with custom turn logic) +│ └─ Nova models via /v1/listen +│ ├─ Pre-recorded file → REST POST https://api.deepgram.com/v1/listen?model=nova-3 +│ └─ Live stream → WSS wss://api.deepgram.com/v1/listen?model=nova-3 +│ +└─ Conversational audio / voice-agent-style turn detection + └─ Flux STT models via /v2/listen + └─ Live stream → WSS wss://api.deepgram.com/v2/listen?model=flux-general-en + +Text → audio (speech synthesis)? +├─ General-purpose TTS (broadest voice catalog, compressed/containerized audio) +│ └─ Aura models via /v1/speak +│ ├─ One-shot → REST POST https://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ └─ Low-latency stream → WSS wss://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ +└─ Voice-agent TTS (turn-based lifecycle, barge-in, cross-turn consistency) + └─ Flux TTS models via /v2/speak — model is REQUIRED, and must be flux-* + ├─ Pre-render a block → REST POST https://api.deepgram.com/v2/speak?model=flux-alexis-en + └─ Live conversation → WSS wss://api.deepgram.com/v2/speak?model=flux-alexis-en + +Full conversational voice agent (audio in, audio out)? +└─ WSS wss://agent.deepgram.com/v1/agent/converse + Deepgram handles STT + your configured LLM + TTS internally + +Analyze text for insights? +└─ REST POST /v1/read + (summaries, sentiment, topics, intents) +``` + +## Speech-to-Text: Nova (`/v1/listen`) vs Flux STT (`/v2/listen`) + +Both model families are actively maintained and industry-leading. They solve different problems — pick the one that matches your use case. + +| | Nova (`/v1/listen`) | Flux STT (`/v2/listen`) | +|---|---|---| +| Endpoint | `/v1/listen` | `/v2/listen` | +| Available models | `nova-3` (also `nova-3-medical`, `nova-3-pharma`), `nova-2`, `nova`, `enhanced`, `base` | `flux-general-en`, `flux-general-multi` | +| Best for | General transcription — captions, subtitles, call logs, batch | Conversational audio — voice agents, interactive assistants, turn-taking UIs | +| Output | Continuous transcript stream | Structured turn events + transcripts (built-in turn state machine) | +| Turn detection | Manual (`utterance_end_ms`, VAD events) | Built-in (EOT, eager-EOT, turn_index) | +| Transports | REST + WebSocket | WebSocket only | +| Intelligence overlays | Yes — `summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`, etc. | No — smaller focused param set; no `smart_format` / `diarize_model` / `punctuate` | +| Mid-session reconfig | No (reconnect to change) | Yes (`Configure` message updates EOT thresholds, keyterms, language hints, and `numerals` live) | + +**Pick Nova (`/v1/listen`, `model=nova-3`) when:** +- Generating captions, subtitles, or transcripts for recorded media +- Running batch transcription over files (REST) +- You need analytics overlays (`summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`) +- You want WebSocket streaming with your own turn-detection logic + +**Pick Flux STT (`/v2/listen`, `model=flux-general-en`) when:** +- Building an interactive voice agent or assistant +- You want end-of-turn detection handled for you +- You need low-latency turn signals and barge-in support +- You want to update EOT thresholds, keyterms, language hints, or `numerals` mid-session without reconnecting + +Migrating from Nova 3 to Flux STT? See the official [Nova 3 → Flux migration guide](https://developers.deepgram.com/docs/flux/nova-3-migration). + +## Text-to-Speech: Aura (`/v1/speak`) vs Flux TTS (`/v2/speak`) + +Both TTS families are actively maintained. `/v2/speak` is a **new endpoint, not a replacement** — `/v1/speak` is unchanged, and there is no aliasing, redirect, or deprecation. The families do not overlap: Aura voices are served only on `/v1/speak`, Flux TTS voices only on `/v2/speak`. + +| | Aura (`/v1/speak`) | Flux TTS (`/v2/speak`) | +|---|---|---| +| Endpoint | `/v1/speak` | `/v2/speak` | +| Models | `aura-2-*` (en, es, de, nl, fr, it, ja), `aura-*` | `flux-{voice}-{language}`, e.g. `flux-alexis-en` — English at launch | +| `model` param | Optional (defaults to `aura-asteria-en`) | **Required**; an `aura-*` string is rejected | +| Best for | Broadest voice catalog, multilingual, compressed audio, one-shot synthesis | Voice agents — streaming LLM output, barge-in, multi-turn conversations | +| Mental model | Text buffer → audio stream | Streaming-first, turn-based conversation | +| Turn lifecycle | None | `SpeechStarted` → audio → `Flushed` → `SpeechMetadata` per turn (server-assigned `speech_id`) | +| Cross-turn context | None (reconnect to reset) | Prosody persists across turns automatically — no API surface | +| Transports | REST + WebSocket | REST (batch) + WebSocket (streaming) | +| Streaming encodings | `linear16`, `mulaw`, `alaw` | `linear16`, `mulaw`, `alaw` — raw audio only | +| Batch encodings | `mp3`, `opus`, `flac`, `aac`, `linear16`, `mulaw`, `alaw` + `container` / `bit_rate` | Same — but batch-only; the socket rejects them | +| Interruption | `Clear` discards the buffer, no feedback | `Interrupt` → `SpeechInterrupted` with `text_spoken` / `text_remaining` | +| Mid-stream reconfig | No (fixed at connection) | Yes — `Configure` updates `speed` only | +| `speed` | `0.7` to `1.5`, Aura-2, English and Spanish only | `0.5` to `1.5` in `0.05` steps; capped at `1.15` when the text carries a pause marker (`PAUSE_SPEED_CAP_EXCEEDED` above that); see Inline controls for the pronunciation rule | +| `expressivity` | Not supported | `-2`…`2`, default `0` (beta; fixed for the connection) | +| Inline controls | Pronunciation `\{"word":"...","pronounce":""\}` (GA on Aura-2, English and Spanish, input up to 2000 characters, combinable with `speed`); no pause control | Pronunciation (Early Access, both transports) only with `speed` exactly `1.0`: `CONTROL_COMBINATION_INVALID` on batch, `DATA-0002` on the socket; pause `\{pause:500ms\}` on batch only, 500 to 3000 ms in 100 ms steps, at most 8 per request | +| Voice Agent `provider.version` | `v1` (the default when a provider is specified) | `v2` (required) | + +**Pick Aura (`/v1/speak`) when:** +- You need a language other than English, or a specific Aura voice +- You need compressed output (`mp3`, `opus`, `flac`, `aac`) inside a Voice Agent, where Flux TTS returns `INVALID_SETTINGS`; on batch REST both families serve those encodings +- You're already on Aura and nothing in Flux TTS is pulling you over — v1 is unchanged + +**Pick Flux TTS (`/v2/speak`) when:** +- Building a voice agent, phone assistant, or customer-service bot +- You're streaming LLM tokens to a speaker in real time and want the lowest time-to-first-audio +- The user may barge in mid-response and you need to know what they actually heard +- You want tone to carry across turns without managing state yourself +- You're pre-rendering fixed audio (IVR prompts, notifications) with a Flux TTS voice — use the batch REST transport + +Migrating from Aura? See the official [Migrating from Aura to Flux TTS](https://developers.deepgram.com/docs/flux-tts/migrating) guide and [Batch vs Streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming). + +## API Domains + +| Domain | REST | WebSocket | Reference | +|--------|------|-----------|-----------| +| Listen v1 — STT, Nova models | `POST /v1/listen` | `wss://api.deepgram.com/v1/listen` | [listen.md](references/listen.md) | +| Listen v2 — STT, Flux STT (conversational) | — | `wss://api.deepgram.com/v2/listen` | [listen.md](references/listen.md) | +| Speak v1 — TTS, Aura models | `POST /v1/speak` | `wss://api.deepgram.com/v1/speak` | [speak.md](references/speak.md) | +| Speak v2 — TTS, Flux TTS (turn-based) | `POST /v2/speak` | `wss://api.deepgram.com/v2/speak` | [speak.md](references/speak.md) | +| Voice Agent | `GET agent.deepgram.com/v1/agent/settings/think/models`; reusable agent configurations at `/v1/projects/{project_id}/agents` (`GET`, `POST`) and `/v1/projects/{project_id}/agents/{agent_id}` (`GET`, `PUT`, `DELETE`); agent variables at `/v1/projects/{project_id}/agent-variables` (`GET`, `POST`) and `/v1/projects/{project_id}/agent-variables/{variable_id}` (`GET`, `PATCH`, `DELETE`) | `wss://agent.deepgram.com/v1/agent/converse` | [agent.md](references/agent.md) | +| Read (Intelligence) | `POST /v1/read` | — | [read.md](references/read.md) | +| Models | `GET /v1/models`, `GET /v1/models/{model_id}`, `GET /v1/projects/{project_id}/models`, `GET /v1/projects/{project_id}/models/{model_id}`; `include_outdated=true` on either list call also returns non-latest model versions | none | [models.md](references/models.md) | +| Projects | `/v1/projects/*` | — | [projects.md](references/projects.md) | +| Auth | `POST /v1/auth/grant` | — | [auth.md](references/auth.md) | +| Self-Hosted | `/v1/projects/*/self-hosted/*` | — | [self-hosted.md](references/self-hosted.md) | + +## Common Mistakes to Avoid + +### All APIs + +1. **Feature flags are query params, except for Voice Agent and the v2 mid-session updates.** For `/v1/listen`, `/v2/listen`, `/v1/speak`, and `/v2/speak`, initial options go on the URL. For Listen, the request body carries audio (REST) or audio frames (WebSocket); for Speak, it carries text (REST JSON `text` field or WebSocket `Speak` messages). Exceptions: `/v1/agent/converse` has no URL query params at all (all config goes in the `Settings` message); `/v2/listen` supports a `Configure` message after connection to update EOT thresholds, keyterms, language hints, and `numerals` mid-session; and `/v2/speak` supports a `Configure` message that updates `speed` only. Also note that `/v2/listen` has a much smaller param set than `/v1/listen`: flags like `smart_format`, `diarize_model`, and `punctuate` are not available. + +2. **Rate limits are concurrent connections, not total requests.** A 429 means too many simultaneous open connections, not too high a request volume. Diarization and other compute-heavy features reduce your concurrency allowance further. Limits apply per project, not per API key, and differ by region; the [API Rate Limits](https://developers.deepgram.com/reference/api-rate-limits) page carries the per-region concurrency tables. + +### STT WebSocket (`/v1/listen`) + +3. **Send KeepAlive as a text frame, not binary.** The connection closes after 10 seconds of no audio. Send `{"type":"KeepAlive"}` as a text (JSON) frame every 3–5 seconds during silence. Sending it as a binary frame causes transcription delays — the audio pipeline chokes — not a silent no-op. + +4. **Never send empty byte payloads.** Sending a zero-length binary frame to `/v1/listen` is treated as a close — it terminates the connection. Always check that your audio packet has length before sending. + +5. **`encoding` must match the actual audio format.** If `encoding=linear16` but you're sending opus, you'll get a DATA-0000 error or garbled output. Omit `encoding` entirely when sending containerized formats (mp3, wav, ogg) — Deepgram detects them automatically. + +6. **Timestamps reset on reconnect.** Each new WebSocket connection restarts timestamps at 00:00:00. For real-time apps, maintain a timestamp offset across reconnections or you'll silently corrupt your transcript timeline. + +### TTS WebSocket (`/v1/speak`) + +7. **Don't send empty text.** A `Speak` message with an empty `text` field returns a 400 error. Always validate input before sending. + +8. **Character rate limiting (DATA-0001) means slow down, not retry.** If you hit this, reduce how fast you're submitting text chunks — don't immediately retry or you'll compound the problem. + +### Flux TTS (`/v2/speak`) + +9. **`model` is required, and must be a `flux-*` voice.** Unlike `/v1/speak` there is no default — a connection or request without `model` is rejected. Aura strings are rejected on `/v2/speak`, and Flux voices are not served by `/v1/speak`; the two families never mix. Model strings are `flux-{voice}-{language}`, e.g. `flux-alexis-en`. There is no version segment — generations roll forward behind a stable name, as with Flux STT. + +10. **`Flush` ends the turn — it is not a v1-style buffer flush.** There is no `Finalize`; it's folded into `Flush`. Audio starts streaming on its own before you flush, so don't wait to send text. Use the turn's `SpeechMetadata` (not `Flushed`) as your end-of-turn signal — it arrives once all of the turn's audio has been sent, and carries the billing and timing counts, so you can drop client-side character or duration tracking. The server assigns the turn's `speech_id`; never send one yourself. + +11. **Streaming is raw audio only, and rejects anything it doesn't recognize.** The WebSocket emits non-containerized audio, so `encoding` is limited to `linear16` (default), `mulaw`, or `alaw`. The compressed and containerized encodings (`mp3`, `opus`, `flac`, `aac`) and the `container`, `bit_rate`, `callback`, `callback_method`, and `priority` params are **batch-only** — sending them to the socket fails the connection, as does any unknown or misspelled param. Use the batch REST transport when you need compressed output. + +12. **Insert whitespace between separate generations, because the server won't.** Text normalization runs before synthesis, but successive `Speak` messages are concatenated verbatim. Sending `"Hello world."` then `"How are you?"` is processed as `"Hello world.How are you?"`, which causes sentence-boundary artifacts. Add a single space (or the right separator for non-whitespace languages) when you stitch a reply, a tool-call result, and another reply together. Send plain text: SSML is not interpreted, and the only markup Flux TTS honors is its own escaped inline controls. A pronunciation override `\{"word":"...","pronounce":""\}` is honored on both transports (Early Access) but only with `speed` 1.0, and a pause marker `\{pause:500ms\}` is batch-only. A pause marker on the socket, or a pronunciation control on a socket whose `speed` is not 1.0, fails the connection with `DATA-0002`. On batch `POST /v2/speak` the same violations are a 400 whose `err_code` names the rule: `CONTROL_COMBINATION_INVALID` (pronunciation with a pause, or with a `speed` other than `1.0`), `PAUSE_SPEED_CAP_EXCEEDED` (a pause marker with `speed` above `1.15`), `BREAK_OUT_OF_RANGE` (a pause outside 500 to 3000 ms), `BREAK_INCREMENT_INVALID` (a pause off the 100 ms grid), `BREAKS_LIMIT_EXCEEDED` (more than 8 pause markers, or two with no text between them), and `BREAK_SYNTAX_INVALID` (a malformed marker, such as a simple marker without backslashes or an escaped structured marker). A `speed` of exactly `1.0` never counts as a speed control, so it triggers none of these. See [Speed, Pause, Pronunciation](https://developers.deepgram.com/docs/tts-voice-controls). + +### Voice Agent (`/v1/agent/converse`) + +13. **Send the `Settings` message before any audio.** The agent ignores everything until it receives and acknowledges the Settings configuration. Message ordering is strictly required. + +14. **`agent.speak.provider.version` selects the TTS family — and omitting `agent.speak` now gives you Flux TTS.** Set `version` to `v2` for Flux TTS or `v1` for Aura; when you specify a provider but omit `version`, it defaults to `v1`. But if you omit `agent.speak` entirely, the agent defaults to Flux TTS with the `flux-kit-en` voice. Switch families by changing `version` and `model` together — a `flux-*` model under `v1`, or an `aura-*` model under `v2`, is invalid: + ```json + { "agent": { "speak": { "provider": { "type": "deepgram", "version": "v2", "model": "flux-alexis-en" } } } } + ``` + +15. **`GET /v1/agent/settings/think/models` lives on `agent.deepgram.com`, not `api.deepgram.com`.** `GET /v1/agent/settings/think/models`, the list of LLMs you can name in `agent.think.provider`, returns **404 on `api.deepgram.com`** and 200 on `agent.deepgram.com`. Same key, same path; only the host differs, so a client with one hardcoded base URL silently gets a 404 that looks like a missing feature. The three regional `api.*` hosts serve it as well. + +### Flux STT model (`/v2/listen`) + +16. **Use `/v2/listen` and a `flux-general-*` model.** Two are served: `flux-general-en` (English) and `flux-general-multi` (multilingual, and the only model that accepts `language_hint` / `language_hints`). `/v1/listen` does not support Flux STT, and `model=flux` alone is not a valid value. Do not include `language` or `encoding` params for containerized audio. + +17. **Use `Configure` to update EOT thresholds, keyterms, language hints, and `numerals` mid-session.** Unlike `/v1/listen`, Flux STT supports live reconfiguration after connection, so there is no need to reconnect to change turn detection sensitivity, boost new keyterms, re-bias language detection (`language_hints`, `flux-general-multi` only), or switch `numerals` on for a PIN or order number: + ```json + { "type": "Configure", "thresholds": { "eot_threshold": 0.8, "eot_timeout_ms": 3000 }, "keyterms": ["Deepgram"] } + ``` + The server responds with `ConfigureSuccess`, which echoes the full active configuration, `numerals` included, not only the fields you sent, or `ConfigureFailure`, which carries `code` and `description` identifying the rejected configuration. Omitted threshold fields keep their current values. + +18. **`ForceEndTurn` outside a turn is a `Warning`, not an error, and the socket stays open.** Sending `{"type":"ForceEndTurn"}` while no turn is in progress returns `{"type":"Warning","code":"FORCE_END_TURN_NO_ACTIVE_TURN","description":"Received ForceEndTurn while no turn was active; the request was ignored."}` and the connection continues. Do not treat it as fatal or reconnect. `references/listen.md` shows the message shape (`ListenV2Warning`: `code`, `description`, `request_id`, `sequence_id`); `code` is a free string there, so the individual codes such as `FORCE_END_TURN_NO_ACTIVE_TURN` come from the [Force End Turn](https://developers.deepgram.com/docs/flux/force-end-turn) docs. When `ForceEndTurn` *does* land mid-turn, the resulting `TurnInfo` carries `event: "EndOfTurn"` with `trigger: "manual"`. `trigger` is `model` | `manual` | `timeout`, it appears on `EndOfTurn` and nowhere else, and it is an open enum, so tolerate values you do not recognize. + +### Nova diarization (`/v1/listen`) + +19. **Use `diarize_model`, and never send it alongside `diarize`.** `diarize` is deprecated. `diarize_model` both enables diarization and picks the version, so you do not also need `diarize=true` — and sending both fails the request: `400 "diarize_model cannot be used together with diarize or diarize_version."`. Values are `latest`, `v1`, and `v2` for batch (`latest` is currently v2), and `latest` or `v1` for streaming. When diarization is on, `metadata.diarize_info` reports which model actually ran (`{"model_uuid": …, "arch": "v2"}`), which is the only way to tell what `latest` resolved to. + +### Text and Audio Intelligence (`/v1/read`, `/v1/listen`) + +20. **`language` is required on `/v1/read`, and it is validated before anything else.** There is no default: omitting it returns `400 INVALID_QUERY_PARAMETER` with the message "Failed to deserialize query parameters: missing field `language`", which masks every other problem in the request. English only: `language=multi` is rejected, and `en-US` is accepted but echoed back as `en`. Two more `/v1/read` shapes worth knowing: the JSON body takes **exactly one** of `text` or `url` (both or neither gives `PAYLOAD_ERROR`, and `url` must point at a plain-text document, since audio gives `REMOTE_CONTENT_ERROR`), and it is POST-only (`GET` and a WebSocket upgrade both return 405). `summarize` on `/v1/read` accepts `v2` as well as `true`. Result paths differ per endpoint: `/v1/read` returns `results.summary.text`, `/v1/listen` returns `results.summary.short`, so code that handles both has to branch. (`sentiment` maps to `results.sentiments` on both.) + +21. **On the Nova streaming socket, only `detect_entities` works — and the other four fail in three different ways.** `detect_entities=true` is supported and puts `entities` at the **top level** of each `Results` message, beside `channel`, not inside `channel.alternatives[0]`. The other four are prerecorded-only: `summarize` fails the handshake with `400 "Summarization is not available for streaming."`; `topics` and `intents` fail it with `403 UNAUTHORIZED_FEATURES_REQUESTED`, which reads like a key-permissions problem even when the same key's prerecorded `topics`/`intents` calls return 200; and `sentiment` is the trap — the handshake succeeds, no error is ever sent, and sentiment simply never appears in the results. + +### Authentication + +22. **JWT TTL applies only to the initial handshake.** Tokens default to 30 seconds. Once the WebSocket connection is established, the token expiring does not close it — tokens are only needed for the upgrade request. + +## SDK-Specific Skills + +This `api` skill covers the product contracts (endpoints, query params, message shapes) that are identical across SDKs. For **language-idiomatic code** — imports, async patterns, builder APIs, common errors — install the SDK-specific skills. Each Deepgram SDK publishes 7 product skills named `deepgram-{lang}-{product}` (e.g. `deepgram-python-speech-to-text`, `deepgram-js-voice-agent`). The `deepgram-{lang}-` prefix avoids collisions when you install skills from multiple SDKs. + +```bash +# Install all skills from a specific SDK +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET + +# Or install a specific product skill from one SDK (note the deepgram-{lang}- prefix) +npx skills add deepgram/deepgram-python-sdk --skill deepgram-python-speech-to-text +npx skills add deepgram/deepgram-js-sdk --skill deepgram-js-voice-agent +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +| Skill | Purpose | +|---|---| +| `recipes` | Minimal runnable snippets per feature per language | +| `examples` | Full integration examples with third-party platforms (Twilio, LiveKit, etc.) | +| `starters` | Runnable starter apps (framework × feature matrix) | +| `docs` | Navigate Deepgram documentation | +| `audio-intelligence` | The `summarize`, `sentiment`, `topics`, `intents`, and `detect_entities` parameters on `/v1/listen` | +| `text-intelligence` | `POST /v1/read` for text you already have | +| `browser-agent` | The Browser Agent SDK packages for running an agent in a browser | +| `cli` | `deepctl` for shell and CI work | +| `self-hosted` | Running Deepgram on your own GPUs | +| `setup-mcp` | Install the Deepgram MCP server | + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) +- [Speech-to-Text Getting Started](https://developers.deepgram.com/docs/stt/getting-started) +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS Models](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Audio Intelligence](https://developers.deepgram.com/docs/audio-intelligence) +- [Self-Hosted Deployments](https://developers.deepgram.com/docs/self-hosted-introduction) +- [Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints) +- [Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints) +- [API Rate Limits](https://developers.deepgram.com/reference/api-rate-limits): per-region concurrency tables for every API; limits apply per project, not per API key +- [Working with Concurrency Rate Limits](https://developers.deepgram.com/docs/working-with-concurrency-rate-limits) diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/deepctl.mdc b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/deepctl.mdc new file mode 100644 index 00000000..b5ec6a76 --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/deepctl.mdc @@ -0,0 +1,996 @@ +--- +name: api +description: > + Deepgram API reference for speech-to-text, text-to-speech, voice agents, audio intelligence, + and account management. Use whenever building with Deepgram APIs — REST or WebSocket. Covers + authentication, all endpoints, query parameters, request/response schemas, and WebSocket + message formats. Reference files are organized by domain: listen (STT — Nova and Flux STT), speak + (TTS — Aura and Flux TTS), agent (voice agents), read (text/audio intelligence), models, + projects, auth, and self-hosted. +--- + +# Deepgram API + +Build with Deepgram's speech-to-text, text-to-speech, voice agent, and audio intelligence APIs. + +> **"Flux" names two separate products.** **Flux STT** is conversational speech-to-text on `/v2/listen` (`model=flux-general-en`). **Flux TTS** is turn-based speech synthesis on `/v2/speak` (`model=flux-{voice}-{language}`). They share a name and a design philosophy — turn-aware, built for voice agents — but they are different endpoints with different models, params, and messages. When a request just says "Flux", check whether it is about transcribing audio or producing it. + +## Getting Started + +All API requests require authentication via API key or JWT: + +- **API Key**: `Authorization: Token ` +- **JWT**: `Authorization: Bearer ` + +Base servers: + +- REST & STT/TTS WebSocket: `https://api.deepgram.com` +- Voice Agent WebSocket **and `GET /v1/agent/settings/think/models`**: `https://agent.deepgram.com` + +`GET /v1/agent/settings/think/models` lives on the `agent.` host too, not on `api.`: it +returns 404 on `api.deepgram.com` and 200 on `agent.deepgram.com`. Everything else REST +stays on `api.deepgram.com`. + +### Regional endpoints + +To keep processing inside a geography, swap the host. Same API keys, same paths, same SDKs — +only the base URL changes. Requests are never routed out of region: if the region is +unavailable they fail rather than fall back. + +| Region | Host | +|---|---| +| EU | `api.eu.deepgram.com` | +| Australia | `api.au.deepgram.com` | +| India | `api.in.deepgram.com` | + +**The data plane is regional; the Projects management API is not.** On all three regional hosts: + +| Endpoint | Regional | +|---|---| +| `POST /v1/listen`, `wss://…/v1/listen` | Yes | +| `wss://…/v2/listen` | Yes | +| `POST /v1/speak`, `wss://…/v1/speak` | Yes | +| `POST /v2/speak`, `wss://…/v2/speak` | Yes | +| `POST /v1/read` | Yes | +| `wss://…/v1/agent/converse` | Yes | +| `GET /v1/models` | Yes | +| `POST /v1/auth/grant` | Yes | +| `/v1/projects/*` (keys, members, usage, billing) | **No — 404** | + +Two host rules that catch people out: + +1. **Voice Agent moves onto the `api.` host regionally.** There is no `agent.eu.deepgram.com` + (the name does not resolve). Use `wss://api.eu.deepgram.com/v1/agent/converse`. `GET /v1/agent/settings/think/models` moves with it. Globally it stays on `agent.deepgram.com`. +2. **Keep management calls on `api.deepgram.com`.** Point a client's management calls at a + regional host and `/v1/projects` returns 404, so split the base URL by call type if your + app both transcribes and manages keys. + +Whisper models are not served in any of the three regions — use Nova or Flux STT models there. + +For Deepgram Dedicated and self-hosted hosts, see +[Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints); +for the full per-region feature matrix and SDK snippets, see +[Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints). + +## How Deepgram's APIs Fit Together + +``` + ┌──────────────────────────────┐ + │ api.deepgram.com │ + └──────────────────────────────┘ + │ + ┌───────────┬───────────┬─────┴─────┬───────────┬───────────┐ + ▼ ▼ ▼ ▼ ▼ ▼ + /v1/listen /v2/listen /v1/speak /v2/speak /v1/read /v1/projects/* + Nova — STT Flux — STT Aura — TTS Flux — TTS Text AI Management + REST + WSS WSS only REST + WSS REST + WSS REST only REST only + + ┌──────────────────────────────┐ + │ agent.deepgram.com │ + └──────────────────────────────┘ + │ + ▼ + /v1/agent/converse + WebSocket only + audio ──▶ STT ──▶ LLM ──▶ TTS ──▶ audio + (Deepgram orchestrates the full pipeline) +``` + +## Which API Should I Use? + +``` +Audio → text (transcription)? +├─ General-purpose transcription (captions, batch, call logs, live streams with custom turn logic) +│ └─ Nova models via /v1/listen +│ ├─ Pre-recorded file → REST POST https://api.deepgram.com/v1/listen?model=nova-3 +│ └─ Live stream → WSS wss://api.deepgram.com/v1/listen?model=nova-3 +│ +└─ Conversational audio / voice-agent-style turn detection + └─ Flux STT models via /v2/listen + └─ Live stream → WSS wss://api.deepgram.com/v2/listen?model=flux-general-en + +Text → audio (speech synthesis)? +├─ General-purpose TTS (broadest voice catalog, compressed/containerized audio) +│ └─ Aura models via /v1/speak +│ ├─ One-shot → REST POST https://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ └─ Low-latency stream → WSS wss://api.deepgram.com/v1/speak?model=aura-2-thalia-en +│ +└─ Voice-agent TTS (turn-based lifecycle, barge-in, cross-turn consistency) + └─ Flux TTS models via /v2/speak — model is REQUIRED, and must be flux-* + ├─ Pre-render a block → REST POST https://api.deepgram.com/v2/speak?model=flux-alexis-en + └─ Live conversation → WSS wss://api.deepgram.com/v2/speak?model=flux-alexis-en + +Full conversational voice agent (audio in, audio out)? +└─ WSS wss://agent.deepgram.com/v1/agent/converse + Deepgram handles STT + your configured LLM + TTS internally + +Analyze text for insights? +└─ REST POST /v1/read + (summaries, sentiment, topics, intents) +``` + +## Speech-to-Text: Nova (`/v1/listen`) vs Flux STT (`/v2/listen`) + +Both model families are actively maintained and industry-leading. They solve different problems — pick the one that matches your use case. + +| | Nova (`/v1/listen`) | Flux STT (`/v2/listen`) | +|---|---|---| +| Endpoint | `/v1/listen` | `/v2/listen` | +| Available models | `nova-3` (also `nova-3-medical`, `nova-3-pharma`), `nova-2`, `nova`, `enhanced`, `base` | `flux-general-en`, `flux-general-multi` | +| Best for | General transcription — captions, subtitles, call logs, batch | Conversational audio — voice agents, interactive assistants, turn-taking UIs | +| Output | Continuous transcript stream | Structured turn events + transcripts (built-in turn state machine) | +| Turn detection | Manual (`utterance_end_ms`, VAD events) | Built-in (EOT, eager-EOT, turn_index) | +| Transports | REST + WebSocket | WebSocket only | +| Intelligence overlays | Yes — `summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`, etc. | No — smaller focused param set; no `smart_format` / `diarize_model` / `punctuate` | +| Mid-session reconfig | No (reconnect to change) | Yes (`Configure` message updates EOT thresholds, keyterms, language hints, and `numerals` live) | + +**Pick Nova (`/v1/listen`, `model=nova-3`) when:** +- Generating captions, subtitles, or transcripts for recorded media +- Running batch transcription over files (REST) +- You need analytics overlays (`summarize`, `sentiment`, `topics`, `intents`, `diarize_model`, `redact`) +- You want WebSocket streaming with your own turn-detection logic + +**Pick Flux STT (`/v2/listen`, `model=flux-general-en`) when:** +- Building an interactive voice agent or assistant +- You want end-of-turn detection handled for you +- You need low-latency turn signals and barge-in support +- You want to update EOT thresholds, keyterms, language hints, or `numerals` mid-session without reconnecting + +Migrating from Nova 3 to Flux STT? See the official [Nova 3 → Flux migration guide](https://developers.deepgram.com/docs/flux/nova-3-migration). + +## Text-to-Speech: Aura (`/v1/speak`) vs Flux TTS (`/v2/speak`) + +Both TTS families are actively maintained. `/v2/speak` is a **new endpoint, not a replacement** — `/v1/speak` is unchanged, and there is no aliasing, redirect, or deprecation. The families do not overlap: Aura voices are served only on `/v1/speak`, Flux TTS voices only on `/v2/speak`. + +| | Aura (`/v1/speak`) | Flux TTS (`/v2/speak`) | +|---|---|---| +| Endpoint | `/v1/speak` | `/v2/speak` | +| Models | `aura-2-*` (en, es, de, nl, fr, it, ja), `aura-*` | `flux-{voice}-{language}`, e.g. `flux-alexis-en` — English at launch | +| `model` param | Optional (defaults to `aura-asteria-en`) | **Required**; an `aura-*` string is rejected | +| Best for | Broadest voice catalog, multilingual, compressed audio, one-shot synthesis | Voice agents — streaming LLM output, barge-in, multi-turn conversations | +| Mental model | Text buffer → audio stream | Streaming-first, turn-based conversation | +| Turn lifecycle | None | `SpeechStarted` → audio → `Flushed` → `SpeechMetadata` per turn (server-assigned `speech_id`) | +| Cross-turn context | None (reconnect to reset) | Prosody persists across turns automatically — no API surface | +| Transports | REST + WebSocket | REST (batch) + WebSocket (streaming) | +| Streaming encodings | `linear16`, `mulaw`, `alaw` | `linear16`, `mulaw`, `alaw` — raw audio only | +| Batch encodings | `mp3`, `opus`, `flac`, `aac`, `linear16`, `mulaw`, `alaw` + `container` / `bit_rate` | Same — but batch-only; the socket rejects them | +| Interruption | `Clear` discards the buffer, no feedback | `Interrupt` → `SpeechInterrupted` with `text_spoken` / `text_remaining` | +| Mid-stream reconfig | No (fixed at connection) | Yes — `Configure` updates `speed` only | +| `speed` | `0.7` to `1.5`, Aura-2, English and Spanish only | `0.5` to `1.5` in `0.05` steps; capped at `1.15` when the text carries a pause marker (`PAUSE_SPEED_CAP_EXCEEDED` above that); see Inline controls for the pronunciation rule | +| `expressivity` | Not supported | `-2`…`2`, default `0` (beta; fixed for the connection) | +| Inline controls | Pronunciation `\{"word":"...","pronounce":""\}` (GA on Aura-2, English and Spanish, input up to 2000 characters, combinable with `speed`); no pause control | Pronunciation (Early Access, both transports) only with `speed` exactly `1.0`: `CONTROL_COMBINATION_INVALID` on batch, `DATA-0002` on the socket; pause `\{pause:500ms\}` on batch only, 500 to 3000 ms in 100 ms steps, at most 8 per request | +| Voice Agent `provider.version` | `v1` (the default when a provider is specified) | `v2` (required) | + +**Pick Aura (`/v1/speak`) when:** +- You need a language other than English, or a specific Aura voice +- You need compressed output (`mp3`, `opus`, `flac`, `aac`) inside a Voice Agent, where Flux TTS returns `INVALID_SETTINGS`; on batch REST both families serve those encodings +- You're already on Aura and nothing in Flux TTS is pulling you over — v1 is unchanged + +**Pick Flux TTS (`/v2/speak`) when:** +- Building a voice agent, phone assistant, or customer-service bot +- You're streaming LLM tokens to a speaker in real time and want the lowest time-to-first-audio +- The user may barge in mid-response and you need to know what they actually heard +- You want tone to carry across turns without managing state yourself +- You're pre-rendering fixed audio (IVR prompts, notifications) with a Flux TTS voice — use the batch REST transport + +Migrating from Aura? See the official [Migrating from Aura to Flux TTS](https://developers.deepgram.com/docs/flux-tts/migrating) guide and [Batch vs Streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming). + +## API Domains + +| Domain | REST | WebSocket | Reference | +|--------|------|-----------|-----------| +| Listen v1 — STT, Nova models | `POST /v1/listen` | `wss://api.deepgram.com/v1/listen` | [listen.md](references/listen.md) | +| Listen v2 — STT, Flux STT (conversational) | — | `wss://api.deepgram.com/v2/listen` | [listen.md](references/listen.md) | +| Speak v1 — TTS, Aura models | `POST /v1/speak` | `wss://api.deepgram.com/v1/speak` | [speak.md](references/speak.md) | +| Speak v2 — TTS, Flux TTS (turn-based) | `POST /v2/speak` | `wss://api.deepgram.com/v2/speak` | [speak.md](references/speak.md) | +| Voice Agent | `GET agent.deepgram.com/v1/agent/settings/think/models`; reusable agent configurations at `/v1/projects/{project_id}/agents` (`GET`, `POST`) and `/v1/projects/{project_id}/agents/{agent_id}` (`GET`, `PUT`, `DELETE`); agent variables at `/v1/projects/{project_id}/agent-variables` (`GET`, `POST`) and `/v1/projects/{project_id}/agent-variables/{variable_id}` (`GET`, `PATCH`, `DELETE`) | `wss://agent.deepgram.com/v1/agent/converse` | [agent.md](references/agent.md) | +| Read (Intelligence) | `POST /v1/read` | — | [read.md](references/read.md) | +| Models | `GET /v1/models`, `GET /v1/models/{model_id}`, `GET /v1/projects/{project_id}/models`, `GET /v1/projects/{project_id}/models/{model_id}`; `include_outdated=true` on either list call also returns non-latest model versions | none | [models.md](references/models.md) | +| Projects | `/v1/projects/*` | — | [projects.md](references/projects.md) | +| Auth | `POST /v1/auth/grant` | — | [auth.md](references/auth.md) | +| Self-Hosted | `/v1/projects/*/self-hosted/*` | — | [self-hosted.md](references/self-hosted.md) | + +## Common Mistakes to Avoid + +### All APIs + +1. **Feature flags are query params, except for Voice Agent and the v2 mid-session updates.** For `/v1/listen`, `/v2/listen`, `/v1/speak`, and `/v2/speak`, initial options go on the URL. For Listen, the request body carries audio (REST) or audio frames (WebSocket); for Speak, it carries text (REST JSON `text` field or WebSocket `Speak` messages). Exceptions: `/v1/agent/converse` has no URL query params at all (all config goes in the `Settings` message); `/v2/listen` supports a `Configure` message after connection to update EOT thresholds, keyterms, language hints, and `numerals` mid-session; and `/v2/speak` supports a `Configure` message that updates `speed` only. Also note that `/v2/listen` has a much smaller param set than `/v1/listen`: flags like `smart_format`, `diarize_model`, and `punctuate` are not available. + +2. **Rate limits are concurrent connections, not total requests.** A 429 means too many simultaneous open connections, not too high a request volume. Diarization and other compute-heavy features reduce your concurrency allowance further. Limits apply per project, not per API key, and differ by region; the [API Rate Limits](https://developers.deepgram.com/reference/api-rate-limits) page carries the per-region concurrency tables. + +### STT WebSocket (`/v1/listen`) + +3. **Send KeepAlive as a text frame, not binary.** The connection closes after 10 seconds of no audio. Send `{"type":"KeepAlive"}` as a text (JSON) frame every 3–5 seconds during silence. Sending it as a binary frame causes transcription delays — the audio pipeline chokes — not a silent no-op. + +4. **Never send empty byte payloads.** Sending a zero-length binary frame to `/v1/listen` is treated as a close — it terminates the connection. Always check that your audio packet has length before sending. + +5. **`encoding` must match the actual audio format.** If `encoding=linear16` but you're sending opus, you'll get a DATA-0000 error or garbled output. Omit `encoding` entirely when sending containerized formats (mp3, wav, ogg) — Deepgram detects them automatically. + +6. **Timestamps reset on reconnect.** Each new WebSocket connection restarts timestamps at 00:00:00. For real-time apps, maintain a timestamp offset across reconnections or you'll silently corrupt your transcript timeline. + +### TTS WebSocket (`/v1/speak`) + +7. **Don't send empty text.** A `Speak` message with an empty `text` field returns a 400 error. Always validate input before sending. + +8. **Character rate limiting (DATA-0001) means slow down, not retry.** If you hit this, reduce how fast you're submitting text chunks — don't immediately retry or you'll compound the problem. + +### Flux TTS (`/v2/speak`) + +9. **`model` is required, and must be a `flux-*` voice.** Unlike `/v1/speak` there is no default — a connection or request without `model` is rejected. Aura strings are rejected on `/v2/speak`, and Flux voices are not served by `/v1/speak`; the two families never mix. Model strings are `flux-{voice}-{language}`, e.g. `flux-alexis-en`. There is no version segment — generations roll forward behind a stable name, as with Flux STT. + +10. **`Flush` ends the turn — it is not a v1-style buffer flush.** There is no `Finalize`; it's folded into `Flush`. Audio starts streaming on its own before you flush, so don't wait to send text. Use the turn's `SpeechMetadata` (not `Flushed`) as your end-of-turn signal — it arrives once all of the turn's audio has been sent, and carries the billing and timing counts, so you can drop client-side character or duration tracking. The server assigns the turn's `speech_id`; never send one yourself. + +11. **Streaming is raw audio only, and rejects anything it doesn't recognize.** The WebSocket emits non-containerized audio, so `encoding` is limited to `linear16` (default), `mulaw`, or `alaw`. The compressed and containerized encodings (`mp3`, `opus`, `flac`, `aac`) and the `container`, `bit_rate`, `callback`, `callback_method`, and `priority` params are **batch-only** — sending them to the socket fails the connection, as does any unknown or misspelled param. Use the batch REST transport when you need compressed output. + +12. **Insert whitespace between separate generations, because the server won't.** Text normalization runs before synthesis, but successive `Speak` messages are concatenated verbatim. Sending `"Hello world."` then `"How are you?"` is processed as `"Hello world.How are you?"`, which causes sentence-boundary artifacts. Add a single space (or the right separator for non-whitespace languages) when you stitch a reply, a tool-call result, and another reply together. Send plain text: SSML is not interpreted, and the only markup Flux TTS honors is its own escaped inline controls. A pronunciation override `\{"word":"...","pronounce":""\}` is honored on both transports (Early Access) but only with `speed` 1.0, and a pause marker `\{pause:500ms\}` is batch-only. A pause marker on the socket, or a pronunciation control on a socket whose `speed` is not 1.0, fails the connection with `DATA-0002`. On batch `POST /v2/speak` the same violations are a 400 whose `err_code` names the rule: `CONTROL_COMBINATION_INVALID` (pronunciation with a pause, or with a `speed` other than `1.0`), `PAUSE_SPEED_CAP_EXCEEDED` (a pause marker with `speed` above `1.15`), `BREAK_OUT_OF_RANGE` (a pause outside 500 to 3000 ms), `BREAK_INCREMENT_INVALID` (a pause off the 100 ms grid), `BREAKS_LIMIT_EXCEEDED` (more than 8 pause markers, or two with no text between them), and `BREAK_SYNTAX_INVALID` (a malformed marker, such as a simple marker without backslashes or an escaped structured marker). A `speed` of exactly `1.0` never counts as a speed control, so it triggers none of these. See [Speed, Pause, Pronunciation](https://developers.deepgram.com/docs/tts-voice-controls). + +### Voice Agent (`/v1/agent/converse`) + +13. **Send the `Settings` message before any audio.** The agent ignores everything until it receives and acknowledges the Settings configuration. Message ordering is strictly required. + +14. **`agent.speak.provider.version` selects the TTS family — and omitting `agent.speak` now gives you Flux TTS.** Set `version` to `v2` for Flux TTS or `v1` for Aura; when you specify a provider but omit `version`, it defaults to `v1`. But if you omit `agent.speak` entirely, the agent defaults to Flux TTS with the `flux-kit-en` voice. Switch families by changing `version` and `model` together — a `flux-*` model under `v1`, or an `aura-*` model under `v2`, is invalid: + ```json + { "agent": { "speak": { "provider": { "type": "deepgram", "version": "v2", "model": "flux-alexis-en" } } } } + ``` + +15. **`GET /v1/agent/settings/think/models` lives on `agent.deepgram.com`, not `api.deepgram.com`.** `GET /v1/agent/settings/think/models`, the list of LLMs you can name in `agent.think.provider`, returns **404 on `api.deepgram.com`** and 200 on `agent.deepgram.com`. Same key, same path; only the host differs, so a client with one hardcoded base URL silently gets a 404 that looks like a missing feature. The three regional `api.*` hosts serve it as well. + +### Flux STT model (`/v2/listen`) + +16. **Use `/v2/listen` and a `flux-general-*` model.** Two are served: `flux-general-en` (English) and `flux-general-multi` (multilingual, and the only model that accepts `language_hint` / `language_hints`). `/v1/listen` does not support Flux STT, and `model=flux` alone is not a valid value. Do not include `language` or `encoding` params for containerized audio. + +17. **Use `Configure` to update EOT thresholds, keyterms, language hints, and `numerals` mid-session.** Unlike `/v1/listen`, Flux STT supports live reconfiguration after connection, so there is no need to reconnect to change turn detection sensitivity, boost new keyterms, re-bias language detection (`language_hints`, `flux-general-multi` only), or switch `numerals` on for a PIN or order number: + ```json + { "type": "Configure", "thresholds": { "eot_threshold": 0.8, "eot_timeout_ms": 3000 }, "keyterms": ["Deepgram"] } + ``` + The server responds with `ConfigureSuccess`, which echoes the full active configuration, `numerals` included, not only the fields you sent, or `ConfigureFailure`, which carries `code` and `description` identifying the rejected configuration. Omitted threshold fields keep their current values. + +18. **`ForceEndTurn` outside a turn is a `Warning`, not an error, and the socket stays open.** Sending `{"type":"ForceEndTurn"}` while no turn is in progress returns `{"type":"Warning","code":"FORCE_END_TURN_NO_ACTIVE_TURN","description":"Received ForceEndTurn while no turn was active; the request was ignored."}` and the connection continues. Do not treat it as fatal or reconnect. `references/listen.md` shows the message shape (`ListenV2Warning`: `code`, `description`, `request_id`, `sequence_id`); `code` is a free string there, so the individual codes such as `FORCE_END_TURN_NO_ACTIVE_TURN` come from the [Force End Turn](https://developers.deepgram.com/docs/flux/force-end-turn) docs. When `ForceEndTurn` *does* land mid-turn, the resulting `TurnInfo` carries `event: "EndOfTurn"` with `trigger: "manual"`. `trigger` is `model` | `manual` | `timeout`, it appears on `EndOfTurn` and nowhere else, and it is an open enum, so tolerate values you do not recognize. + +### Nova diarization (`/v1/listen`) + +19. **Use `diarize_model`, and never send it alongside `diarize`.** `diarize` is deprecated. `diarize_model` both enables diarization and picks the version, so you do not also need `diarize=true` — and sending both fails the request: `400 "diarize_model cannot be used together with diarize or diarize_version."`. Values are `latest`, `v1`, and `v2` for batch (`latest` is currently v2), and `latest` or `v1` for streaming. When diarization is on, `metadata.diarize_info` reports which model actually ran (`{"model_uuid": …, "arch": "v2"}`), which is the only way to tell what `latest` resolved to. + +### Text and Audio Intelligence (`/v1/read`, `/v1/listen`) + +20. **`language` is required on `/v1/read`, and it is validated before anything else.** There is no default: omitting it returns `400 INVALID_QUERY_PARAMETER` with the message "Failed to deserialize query parameters: missing field `language`", which masks every other problem in the request. English only: `language=multi` is rejected, and `en-US` is accepted but echoed back as `en`. Two more `/v1/read` shapes worth knowing: the JSON body takes **exactly one** of `text` or `url` (both or neither gives `PAYLOAD_ERROR`, and `url` must point at a plain-text document, since audio gives `REMOTE_CONTENT_ERROR`), and it is POST-only (`GET` and a WebSocket upgrade both return 405). `summarize` on `/v1/read` accepts `v2` as well as `true`. Result paths differ per endpoint: `/v1/read` returns `results.summary.text`, `/v1/listen` returns `results.summary.short`, so code that handles both has to branch. (`sentiment` maps to `results.sentiments` on both.) + +21. **On the Nova streaming socket, only `detect_entities` works — and the other four fail in three different ways.** `detect_entities=true` is supported and puts `entities` at the **top level** of each `Results` message, beside `channel`, not inside `channel.alternatives[0]`. The other four are prerecorded-only: `summarize` fails the handshake with `400 "Summarization is not available for streaming."`; `topics` and `intents` fail it with `403 UNAUTHORIZED_FEATURES_REQUESTED`, which reads like a key-permissions problem even when the same key's prerecorded `topics`/`intents` calls return 200; and `sentiment` is the trap — the handshake succeeds, no error is ever sent, and sentiment simply never appears in the results. + +### Authentication + +22. **JWT TTL applies only to the initial handshake.** Tokens default to 30 seconds. Once the WebSocket connection is established, the token expiring does not close it — tokens are only needed for the upgrade request. + +## SDK-Specific Skills + +This `api` skill covers the product contracts (endpoints, query params, message shapes) that are identical across SDKs. For **language-idiomatic code** — imports, async patterns, builder APIs, common errors — install the SDK-specific skills. Each Deepgram SDK publishes 7 product skills named `deepgram-{lang}-{product}` (e.g. `deepgram-python-speech-to-text`, `deepgram-js-voice-agent`). The `deepgram-{lang}-` prefix avoids collisions when you install skills from multiple SDKs. + +```bash +# Install all skills from a specific SDK +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET + +# Or install a specific product skill from one SDK (note the deepgram-{lang}- prefix) +npx skills add deepgram/deepgram-python-sdk --skill deepgram-python-speech-to-text +npx skills add deepgram/deepgram-js-sdk --skill deepgram-js-voice-agent +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +| Skill | Purpose | +|---|---| +| `recipes` | Minimal runnable snippets per feature per language | +| `examples` | Full integration examples with third-party platforms (Twilio, LiveKit, etc.) | +| `starters` | Runnable starter apps (framework × feature matrix) | +| `docs` | Navigate Deepgram documentation | +| `audio-intelligence` | The `summarize`, `sentiment`, `topics`, `intents`, and `detect_entities` parameters on `/v1/listen` | +| `text-intelligence` | `POST /v1/read` for text you already have | +| `browser-agent` | The Browser Agent SDK packages for running an agent in a browser | +| `cli` | `deepctl` for shell and CI work | +| `self-hosted` | Running Deepgram on your own GPUs | +| `setup-mcp` | Install the Deepgram MCP server | + +## Documentation + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) +- [Speech-to-Text Getting Started](https://developers.deepgram.com/docs/stt/getting-started) +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS Models](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Audio Intelligence](https://developers.deepgram.com/docs/audio-intelligence) +- [Self-Hosted Deployments](https://developers.deepgram.com/docs/self-hosted-introduction) +- [Regional Endpoints](https://developers.deepgram.com/reference/regional-endpoints) +- [Custom Endpoints](https://developers.deepgram.com/reference/custom-endpoints) +- [API Rate Limits](https://developers.deepgram.com/reference/api-rate-limits): per-region concurrency tables for every API; limits apply per project, not per API key +- [Working with Concurrency Rate Limits](https://developers.deepgram.com/docs/working-with-concurrency-rate-limits) + + +--- + +--- +name: docs +description: > + Find the right Deepgram documentation for any task. Use whenever someone needs help locating + docs, understanding which API to use, or wants to ask questions about Deepgram. Covers all + product areas: speech-to-text (Nova, Flux STT), text-to-speech (Aura, Flux TTS), voice agents, + audio intelligence, and self-hosted deployments. +--- + +# Deepgram Documentation + +Find the right docs for what you're building with Deepgram. + +## Ask AI + +Have a question? Get answers from Deepgram's AI assistant at . + +## Documentation by Topic + +### Speech-to-Text (STT) + +Transcribe audio and video into text. Deepgram ships two actively maintained, next-gen model families — pick the one that matches your use case. + +- **Nova** (`/v1/listen`) — general-purpose transcription (captions, subtitles, batch files, live streams). Rich feature set including intelligence overlays (diarize, summarize, sentiment, topics, intents). +- **Flux STT** (`/v2/listen`) — conversational-audio transcription for voice agents and interactive assistants. Built-in turn-taking (EOT events, mid-session reconfig). + +Docs: +- [STT Getting Started (Nova)](https://developers.deepgram.com/docs/stt/getting-started) +- [Flux STT Quickstart](https://developers.deepgram.com/docs/flux/quickstart) +- [Nova 3 → Flux STT migration](https://developers.deepgram.com/docs/flux/nova-3-migration) +- [Flux STT language prompting](https://developers.deepgram.com/docs/flux/language-prompting) + +### Text-to-Speech (TTS) + +Convert text into natural-sounding speech. Deepgram ships two TTS model families on separate endpoints — the voices do not overlap. + +- **Aura** (`/v1/speak`) — the broadest voice catalog (English, Spanish, German, Dutch, French, Italian, Japanese) and compressed/containerized output. Use for one-shot synthesis and any non-English voice. +- **Flux TTS** (`/v2/speak`) — streaming-first, voice-agent-first synthesis. Turn-based lifecycle, barge-in with spoken-text feedback, and prosody that carries across turns. English at launch. + +Docs: +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Aura voices and languages](https://developers.deepgram.com/docs/tts-models) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Flux TTS Streaming Quickstart](https://developers.deepgram.com/docs/flux-tts/quickstart) +- [Flux TTS Batch (REST) Quickstart](https://developers.deepgram.com/docs/flux-tts/batch) +- [Flux TTS batch vs streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming) +- [Flux TTS voices](https://developers.deepgram.com/docs/flux-tts/voices) +- [Aura → Flux TTS migration](https://developers.deepgram.com/docs/flux-tts/migrating) + +### Voice Agent + +Build conversational voice agents powered by Deepgram. + +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS models (Aura vs Flux TTS)](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Build a Flux TTS voice agent](https://developers.deepgram.com/docs/flux-tts/voice-agent) + +### Text and Audio Intelligence + +Analyze text and audio for sentiment, topics, intents, summaries, and more. + +- [Audio Intelligence Docs](https://developers.deepgram.com/docs/audio-intelligence) + +### Self-Hosted Deployments + +Run Deepgram on your own infrastructure. + +- [Self-Hosted Introduction](https://developers.deepgram.com/docs/self-hosted-introduction) + +### API Reference + +Full reference for all Deepgram REST and WebSocket APIs. + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## SDK-Specific Skills + +For language-idiomatic code patterns (imports, async idioms, error handling, type shapes), install the Deepgram SDK's own skills. Every Deepgram SDK publishes 7 product skills: + +```bash +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +- `api`: consolidated REST + WebSocket API reference +- `recipes`: minimal runnable feature snippets per language +- `examples`: full integration examples with third-party platforms +- `starters`: runnable starter apps (framework × feature) +- `audio-intelligence`: the `/v1/listen` analysis parameters +- `text-intelligence`: `POST /v1/read` for text you already have +- `browser-agent`: running a voice agent in a browser +- `cli`: `deepctl` for shell and CI work +- `self-hosted`: running Deepgram on your own GPUs +- `setup-mcp`: Deepgram MCP server installation + +## MCP Server + +For direct documentation querying from your AI coding tool, use the `setup-mcp` skill to install the Deepgram MCP server. + + +--- + +--- +name: setup-mcp +description: > + Set up a Deepgram MCP server for your AI coding tool. Offers three paths: the Deepgram CLI + MCP proxy (dg mcp), the standalone deepgram-mcp package, and the credential-free hosted + documentation MCP. Use whenever someone wants to install Deepgram's agentic tools, set up + the MCP server, or connect their editor to Deepgram. +--- + +# Install a Deepgram MCP Server + +You are setting up Deepgram MCP integration for the user. Follow these steps. + +## Step 1: Pick a path + +Three paths exist. Pick by whether the user has, or wants, a Deepgram API key. + +| Path | Server | Credentials | Install footprint | +|---|---|---|---| +| **A** | Deepgram CLI MCP proxy (`dg mcp`) | Deepgram API key **required** | Full CLI (`deepctl`) | +| **B** | Standalone `deepgram-mcp` | Deepgram API key **required** | One Python package | +| **C** | Hosted docs MCP (`/_mcp/server`) | **None** | Nothing to install | + +Decision rule: + +- The user already has the CLI, or wants `dg listen` / `dg speak` / `dg init` too → **Path A**. +- The user has an API key but wants only the MCP server, no CLI → **Path B**. +- The user has no API key, or wants something working in one command → **Path C**. + +A key-authenticated hosted variant of Paths A/B also exists at `api.dx.deepgram.com/kapa/mcp`, +with nothing to install — see "The kapa endpoints are not credential-free" below. + +Paths A and B are the same server: `dg mcp` wraps the `deepgram-mcp` package. Both proxy +Deepgram's developer API and fetch their tool list from Deepgram at runtime, so new tools +appear on reconnect without a package upgrade. As of this writing that list is a single +documentation and knowledge-source search tool (`search_deepgram_knowledge_sources`) — check +`tools/list` in the user's client for what is live rather than promising a tool set. + +Paths A/B and Path C both answer Deepgram questions from documentation, so installing more +than one is usually redundant. Path C is the only one that works with no credentials. + +## Step 2: Detect the environment + +Determine which AI coding tool the user is running. Check for: + +- **Claude Code** — look for a `.claude/` directory in the project or user home +- **Cursor** — look for a `.cursor/` directory in the project root +- **Windsurf** — look for a `.windsurf/` directory in the project root + +If multiple are detected, or none are detected, ask the user which tool they want to configure. + +## Step 3: Ask about scope + +Ask the user whether they want the MCP server configured: + +- **For this project only** (recommended for team repos) +- **Globally** (available in all projects) + +--- + +## Path A — Deepgram CLI MCP proxy (`dg mcp`) + +### A1. Install the CLI + +Check first: `dg --version` (or `deepctl --version`, or `where dg` on Windows). The package is +`deepctl` and installs three interchangeable binaries — `dg`, `deepctl`, and `deepgram`. + +```sh +# macOS / Linux — Homebrew (also brings in ffmpeg and portaudio) +brew install deepgram/tap/deepgram + +# macOS / Linux — install script +curl -fsSL https://deepgram.com/install.sh | sh + +# pip / uv / pipx +pip install deepctl +uv tool install deepctl +pipx install deepctl +``` + +```powershell +# Windows — PowerShell +iwr https://deepgram.com/install.ps1 -useb | iex +``` + +To upgrade, use the installer that put it there: `pip install -U deepctl`, +`uv tool upgrade deepctl`, `pipx upgrade deepctl`, `brew upgrade deepgram`, or re-run the install +script. `dg update --check-only` reports whether a newer release exists; on a pip install, bare +`dg update` reports `installation_method: null` instead of upgrading. + +The fully qualified Homebrew name matters: Homebrew 6 loads a third-party formula only after it +is trusted, and `brew install deepgram/tap/deepgram` trusts that one formula, where +`brew tap deepgram/tap && brew install deepgram` fails until a separate `brew trust` step. The +tap formula pins `deepctl-0.2.26`; pip, uv, and pipx install 0.3.1. + +### A2. Authenticate — required + +`dg mcp` will not start without credentials. Do this before configuring any editor: + +```sh +dg login # interactive; or dg login --api-key +dg whoami # confirm: "authenticated": true +``` + +`DEEPGRAM_API_KEY` in the environment works instead of `dg login`. Get a key at +. + +### A3. Configure the editor + +#### Claude Code + +```sh +# Project scope +claude mcp add deepgram --scope project dg mcp + +# User/global scope +claude mcp add deepgram dg mcp +``` + +#### Cursor + +Write or merge into the project's `.cursor/mcp.json`: + +```json +{ + "mcpServers": { + "deepgram": { + "type": "stdio", + "command": "dg", + "args": ["mcp"] + } + } +} +``` + +#### Windsurf + +Write or merge into the project's `.windsurf/mcp.json`, using the same object as Cursor above. + +#### Without a permanent install + +`uvx` and `pipx run` fetch `deepctl` on demand. Credentials still come from `dg login` or +`DEEPGRAM_API_KEY`: + +```json +{ + "mcpServers": { + "deepgram": { + "command": "uvx", + "args": ["deepctl", "mcp"] + } + } +} +``` + +#### Other tools + +- **Transport:** stdio +- **Command:** `dg` +- **Args:** `["mcp"]` + +`dg mcp --transport sse --port 8000` serves SSE instead, for clients that need HTTP. + +--- + +## Path B — Standalone `deepgram-mcp` + +The MCP server without the rest of the CLI. One package, one binary. + +```sh +pip install deepgram-mcp +export DEEPGRAM_API_KEY=your_key_here +``` + +`deepgram-mcp` is a PyPI package. The npm package of the same name is unrelated third-party code +that also asks for `DEEPGRAM_API_KEY`, so do not run `npx deepgram-mcp`. + +#### Claude Code + +```sh +claude mcp add deepgram -- deepgram-mcp +``` + +#### Cursor / Windsurf + +Write or merge into `.cursor/mcp.json` or the Windsurf MCP config: + +```json +{ + "mcpServers": { + "deepgram": { + "command": "deepgram-mcp", + "env": { + "DEEPGRAM_API_KEY": "your_key_here" + } + } + } +} +``` + +`--api-key` overrides the environment variable, and `--transport sse --port 8000` serves SSE. +Source: . + +--- + +## Path C — Hosted documentation MCP (no credentials) + +Use `https://developers.deepgram.com/_mcp/server`. It answers unauthenticated, needs no API +key, and exposes one tool, `searchDocs`, which returns documentation passages with source URLs. + +It is not a plain liveness URL. `HEAD` returns 404, a `GET` with the MCP +`Accept: application/json, text/event-stream` header returns 405, and a bare `GET` returns a +JSON descriptor of the server rather than an MCP response. Only a `POST` `initialize` exercises +the server; it answers 200 with `serverInfo.name` `fern-docs-mcp-server`: + +```sh +curl -s -X POST https://developers.deepgram.com/_mcp/server \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-03-26","capabilities":{},"clientInfo":{"name":"probe","version":"0"}}}' +``` + +#### Claude Code + +```sh +# Project scope +claude mcp add deepgram-docs --scope project --transport http https://developers.deepgram.com/_mcp/server + +# User/global scope +claude mcp add deepgram-docs --transport http https://developers.deepgram.com/_mcp/server +``` + +#### Cursor + +Write or merge into the project's `.cursor/mcp.json`: + +```json +{ + "mcpServers": { + "deepgram-docs": { + "type": "http", + "url": "https://developers.deepgram.com/_mcp/server" + } + } +} +``` + +#### Windsurf + +Write or merge into the project's `.windsurf/mcp.json`, using the same object as Cursor above. + +#### Other tools + +- **Type:** HTTP +- **URL:** `https://developers.deepgram.com/_mcp/server` + +### The kapa endpoints are not credential-free + +`https://api.dx.deepgram.com/kapa/mcp` and `https://deepgram.mcp.kapa.ai` both exist, and both +reject an unauthenticated request with HTTP 401 plus a `WWW-Authenticate: Bearer +resource_metadata=...` header, so a client that implements MCP's OAuth flow can connect to either. +They differ in whether a Deepgram API key works: + +- **`api.dx.deepgram.com/kapa/mcp` accepts a Deepgram API key.** Send it as either + `Authorization: Token ` or `Authorization: Bearer ` and `initialize` returns 200 from + `deepgram-mcp-relay`; an invalid key gets 401. `tools/list` returns the same single + `search_deepgram_knowledge_sources` tool as Paths A and B, so this is the hosted HTTP form of + the same server — useful when the user has a key but cannot install anything. Pass the key as a + header, or the client falls back to OAuth: + + ```sh + claude mcp add deepgram-relay --transport http https://api.dx.deepgram.com/kapa/mcp \ + --header "Authorization: Token $DEEPGRAM_API_KEY" + ``` +- **`deepgram.mcp.kapa.ai` does not.** A Deepgram API key gets 401 with either scheme. OAuth is + the only way in. + +Neither is the zero-setup option — use `/_mcp/server` for that. + +--- + +## Step 4: Confirm + +- **Claude Code** — run `/reload-plugins` to activate immediately, no restart needed. +- **Cursor / Windsurf / Other** — the user may need to restart or reload their tool. + +Then tell the user the server is configured, and check what it actually exposes before +describing it — have the client list its tools rather than naming tools from memory. + +For Path C, add: + +> Your tool can now search Deepgram's documentation directly — try asking about API +> parameters, voice agents, or model capabilities. + +Link them to [Deepgram Agentic Tools](https://developers.deepgram.com/developer-tools/agentic-tools) +for more details. Its two kapa URLs, `https://api.dx.deepgram.com/kapa/mcp` and +`https://deepgram.mcp.kapa.ai`, require credentials: an unauthenticated `initialize` returns 401. +The Docs MCP server at `https://developers.deepgram.com/_mcp/server` is the credential-free path. + +## Troubleshooting + +**`Error: DEEPGRAM_API_KEY is not set in the configuration file (...config.yaml) or environment variable.`** +followed by `Run deepctl login to configure the CLI with your Deepgram account.` +→ Path A with no credentials. `dg mcp` exits 1 before serving anything. Run `dg login`, or set +`DEEPGRAM_API_KEY`. Confirm with `dg whoami`. + +**`Error: No API key. Set DEEPGRAM_API_KEY or use --api-key.`** +→ Path B with no credentials. Export `DEEPGRAM_API_KEY`, put it in the server's `env` block, or +pass `--api-key`. + +**`! Needs authentication` in `claude mcp list`, or HTTP 401 `{"status_code":401,"detail":"Authentication required"}` / `{"error":"invalid_token"}`** +→ You are pointed at a kapa endpoint with no credentials. Switch to +`https://developers.deepgram.com/_mcp/server`, which needs none. To stay on +`api.dx.deepgram.com/kapa/mcp`, add `--header "Authorization: Token $DEEPGRAM_API_KEY"` — that +endpoint accepts a Deepgram API key. On `deepgram.mcp.kapa.ai` an API key does not work; let the +client run its OAuth flow instead. + +**`Server "deepgram-docs" is defined in multiple scopes with different endpoints`** +→ An earlier setup registered `deepgram-docs` at a kapa URL in user scope, and this one added a +different URL in project scope. OAuth tokens are stored per endpoint, so authenticating one does +not carry over. Keep one: `claude mcp remove deepgram-docs -s user` (or `-s project`). Check for +a pre-existing entry with `claude mcp get deepgram-docs` before adding, and pick a distinct +server name if the user wants to keep both. + +**`ImportError` mentioning `streamablehttp_client` on startup** +→ An incompatible `mcp` package. `deepgram-mcp` imports `streamablehttp_client` from +`mcp.client.streamable_http`, which `mcp` 2.0 removed. Install into a clean environment, or pin +`mcp>=1.0.0,<2.0.0`. Installing `deepctl` pins this for you. + +**The server connects but exposes fewer tools than expected** +→ Expected. Paths A and B fetch their tool list from Deepgram at runtime, so it reflects what +the API serves right now, not what the package version implies. Reconnect to pick up new tools. + +**Anything else on Path A** +→ Verify `dg --version` works and `dg mcp` runs in a terminal without errors, then +`dg update --check-only` to see whether a newer release exists. + +## Sources + +- Deepgram CLI: +- `deepgram-mcp`: +- Deepgram Agentic Tools: + + +--- + +--- +name: starters +description: > + Clone a ready-to-run Deepgram demo app and start building on top of it. Use whenever someone + wants a quick working demo, needs to prototype with Deepgram, or is starting a new project + that uses speech-to-text, text-to-speech, voice agents, audio intelligence, or live streaming. + Match the user's language, framework, and desired Deepgram feature to the right starter. +--- + +# Deepgram Starter Apps + +Clone a working demo and start building. Every starter is a minimal, runnable app you can extend. + +## 1. Pick Your Feature + +What do you want to build? + +- **Transcribe a file** → `transcription` — send audio/video, get text back (REST, Nova) +- **Transcribe a live stream** → `live-transcription` — real-time speech-to-text (WebSocket, Nova) +- **Generate speech** → `text-to-speech` — send text, get audio back (REST, Aura) +- **Stream speech** → `live-text-to-speech` — real-time text-to-audio (WebSocket, Aura) +- **Analyze text** → `text-intelligence` — sentiment, topics, intents, summaries over text you + already have (REST, `/v1/read`) +- **Build a voice agent** → `voice-agent` — conversational AI agent (WebSocket, agent.deepgram.com) +- **Conversational STT with turn detection** → `flux` — Deepgram Flux STT for voice agents and interactive assistants (WebSocket, `/v2/listen`) +- **Turn-based TTS for a voice agent** → `flux-tts` — Deepgram Flux TTS, streaming synthesis with barge-in (WebSocket, `/v2/speak`) + +**There is no audio-intelligence starter.** `text-intelligence` is text-only — it posts text you +already have to `/v1/read`. No `{framework}-audio-intelligence` repository exists in +`deepgram-starters` for any framework, so don't construct those URLs. To run intelligence features +(summarization, sentiment, topics, intents) over *audio*, they are query parameters on +`/v1/listen`, not a separate starter: clone the `transcription` starter for your framework and add +the parameters to its existing request. See the `api` skill for which features `/v1/listen` +supports. + +**Nova vs Flux STT for speech-to-text:** use `transcription` or `live-transcription` (Nova, `/v1/listen`) for general-purpose transcription, captions, and batch workloads. Use `flux` (Flux STT, `/v2/listen`) when you need built-in turn detection for conversational audio. See the `api` skill for a full comparison. + +**Aura vs Flux TTS for text-to-speech:** use `text-to-speech` or `live-text-to-speech` (Aura, `/v1/speak`) for one-shot synthesis, non-English voices, and compressed audio. Use `flux-tts` (Flux TTS, `/v2/speak`) when you're streaming LLM output to a speaker and need a turn lifecycle and barge-in. See the `api` skill for a full comparison. + +**Flux TTS starters exist for `node`, `flask`, `fastapi`, `django`, and `java` only** — these are the five apps Deepgram officially publishes at [Flux TTS template apps](https://developers.deepgram.com/docs/flux-tts/template-apps). There is no `flux-tts` starter for the other frameworks; don't construct those URLs. For an unsupported framework, start from the `api` skill's Flux TTS section and the SDK skills instead. + +## 2. Pick Your Stack + +| Language | Frameworks | +|----------|------------| +| JavaScript | `node` | +| TypeScript | `bun`, `deno` | +| Python | `fastapi`, `flask`, `django` | +| Go | `go` | +| Java | `java` | +| C# | `csharp` | +| Rust | `rust` | +| Ruby | `ruby` | +| PHP | `php` | +| C++ | `cpp` | + +## 3. Clone and Run + +Every starter lives at `https://github.com/deepgram-starters/{framework}-{feature}` — framework +first, feature second. Clone **with submodules**; each starter vendors two git submodules — its +browser frontend at `frontend/` and the shared starter contracts at `contracts/` — and a plain +`git clone` leaves both directories empty and the app unrunnable: + +```sh +git clone --recurse-submodules https://github.com/deepgram-starters/{framework}-{feature}.git +cd {framework}-{feature} +``` + +In 80 of the 96 starters, both submodule URLs in `.gitmodules` are SSH (`git@github.com:...`) +even though both repositories are public, so `--recurse-submodules` fails with +`Host key verification failed` unless the user has a GitHub SSH key. The other 16 use HTTPS URLs +and clone without a key: 12 of the 13 `{framework}-live-transcription` starters (every one except +`rust-live-transcription`) plus `csharp-voice-agent`, `django-voice-agent`, `flask-voice-agent`, +and `node-voice-agent`. Without an SSH key, rewrite SSH to HTTPS for the clone. The rewrite +changes nothing on the 16 HTTPS starters, so it is safe to use on every starter: + +```sh +git -c url."https://github.com/".insteadOf="git@github.com:" \ + clone --recurse-submodules https://github.com/deepgram-starters/{framework}-{feature}.git +``` + +The starter's own `make init` runs `git submodule update --init --recursive` and installs +dependencies, but it inherits the URLs in `.gitmodules`. On the 80 SSH starters it fails +identically without a key, so it is the path for users who **have** SSH set up (or for one of the +16 HTTPS starters), not a workaround for users who don't. + +Set your API key and follow the README: + +```sh +export DEEPGRAM_API_KEY=your_key_here +``` + +Get an API key at . + +### Or scaffold with the CLI + +The [Deepgram CLI](https://github.com/deepgram/cli) has a scaffolder that finds and clones a +starter for you: + +```sh +dg init --list # browse templates +dg init --list --search python # filter +dg init node-transcription # clone into ./node-transcription +dg init node-transcription --dir ./my-app +``` + +**`dg init` does not solve the submodule problem.** It runs a plain clone, so `frontend/` and +`contracts/` land empty, and it still prints `Done! … is ready` and `"status": "success"`. Adding +`--install` runs the starter's `make check-prereqs && make init`, which hits the same `.gitmodules` +URLs: on the 80 SSH starters it fails with `Host key verification failed`, and `dg init` reports +success anyway. Without a GitHub SSH key, finish the checkout by hand after `dg init`: + +```sh +cd my-app +git -c url."https://github.com/".insteadOf="git@github.com:" \ + submodule update --init --recursive +``` + +`dg init` is also marked alpha, and its templates gallery is a separate list from the matrix +below rather than a subset of it. It carries 44 templates with no `flux` or `flux-tts` entries; +it still lists `sinatra-transcription`, whose repository is archived and private, so the clone +returns 404 for anyone outside Deepgram; and it lists `nextjs-*` templates that now redirect out +of `deepgram-starters` to `deepgram-devs`, which is why there is no `nextjs` row below. Treat +the matrix as authoritative and fall back to `git clone`. See the `cli` skill for installing +`deepctl` and for the rest of `dg init`. + +## The `{feature}-html` repos are not starters + +The `deepgram-starters` org also contains `transcription-html`, `live-transcription-html`, +`text-to-speech-html`, `live-text-to-speech-html`, `text-intelligence-html`, `voice-agent-html`, +`flux-html`, and `flux-tts-html`. **Do not clone these and do not offer them as starters.** Each +is the shared browser frontend that a backend starter pulls in as its `frontend/` submodule — +`node-transcription` vendors `transcription-html`, `flask-voice-agent` vendors `voice-agent-html`, +`node-flux-tts` and `java-flux-tts` both vendor `flux-tts-html`, and so on. Seven of the eight +say so in their own README ("This is a frontend submodule - do not use directly"); `flux-tts-html` +carries no such warning but is vendored the same way. None of them serve an API, so none of them +run standalone. Clone the backend starter instead and the right frontend arrives with it. + +They also invert the naming rule. The starter pattern is `{framework}-{feature}`, but these are +`{feature}-html` — and the mirror-image names do **not** exist, so do not construct them: +`deepgram-starters/html-transcription` is a 404. There is no vanilla-HTML row in the matrix +because there is no standalone browser starter; for browser-only work, clone the `node` starter +for the feature you want and read its `frontend/` directory. + +## Examples + +**"I want to build a voice agent in Python"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/fastapi-voice-agent.git` + +**"I need live transcription in my Node app"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/node-live-transcription.git` + +**"I want to add text-to-speech to my Go service"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/go-text-to-speech.git` + +**"I want to analyze audio for sentiment in C#"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/csharp-text-intelligence.git` + +**"I want streaming TTS with barge-in for my Node voice agent"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/node-flux-tts.git` + +**"I want a plain browser/HTML demo"** +→ There is no standalone HTML starter. Clone `node-{feature}` and work in its `frontend/` +directory — that is the same browser code the `{feature}-html` submodule holds. + +## All Starters + +Every URL below is a real, published, non-archived repository, and the table is the complete +set: 13 frameworks × 7 features, plus `flux-tts` for the five frameworks that have it. A cell +showing `—` means that starter does not exist; don't construct the URL. + +The `java-flux-tts` README clones with a plain `git clone`, without `--recurse-submodules`, while +its `.gitmodules` points both submodules at SSH URLs, so following its Maven steps leaves +`frontend/` and `contracts/` empty. Use the clone command in section 3 instead. + +| | transcription | live-transcription | text-to-speech | live-text-to-speech | text-intelligence | voice-agent | flux | flux-tts | +|---|---|---|---|---|---|---|---|---| +| **node** | [repo](https://github.com/deepgram-starters/node-transcription) | [repo](https://github.com/deepgram-starters/node-live-transcription) | [repo](https://github.com/deepgram-starters/node-text-to-speech) | [repo](https://github.com/deepgram-starters/node-live-text-to-speech) | [repo](https://github.com/deepgram-starters/node-text-intelligence) | [repo](https://github.com/deepgram-starters/node-voice-agent) | [repo](https://github.com/deepgram-starters/node-flux) | [repo](https://github.com/deepgram-starters/node-flux-tts) | +| **bun** | [repo](https://github.com/deepgram-starters/bun-transcription) | [repo](https://github.com/deepgram-starters/bun-live-transcription) | [repo](https://github.com/deepgram-starters/bun-text-to-speech) | [repo](https://github.com/deepgram-starters/bun-live-text-to-speech) | [repo](https://github.com/deepgram-starters/bun-text-intelligence) | [repo](https://github.com/deepgram-starters/bun-voice-agent) | [repo](https://github.com/deepgram-starters/bun-flux) | — | +| **deno** | [repo](https://github.com/deepgram-starters/deno-transcription) | [repo](https://github.com/deepgram-starters/deno-live-transcription) | [repo](https://github.com/deepgram-starters/deno-text-to-speech) | [repo](https://github.com/deepgram-starters/deno-live-text-to-speech) | [repo](https://github.com/deepgram-starters/deno-text-intelligence) | [repo](https://github.com/deepgram-starters/deno-voice-agent) | [repo](https://github.com/deepgram-starters/deno-flux) | — | +| **fastapi** | [repo](https://github.com/deepgram-starters/fastapi-transcription) | [repo](https://github.com/deepgram-starters/fastapi-live-transcription) | [repo](https://github.com/deepgram-starters/fastapi-text-to-speech) | [repo](https://github.com/deepgram-starters/fastapi-live-text-to-speech) | [repo](https://github.com/deepgram-starters/fastapi-text-intelligence) | [repo](https://github.com/deepgram-starters/fastapi-voice-agent) | [repo](https://github.com/deepgram-starters/fastapi-flux) | [repo](https://github.com/deepgram-starters/fastapi-flux-tts) | +| **flask** | [repo](https://github.com/deepgram-starters/flask-transcription) | [repo](https://github.com/deepgram-starters/flask-live-transcription) | [repo](https://github.com/deepgram-starters/flask-text-to-speech) | [repo](https://github.com/deepgram-starters/flask-live-text-to-speech) | [repo](https://github.com/deepgram-starters/flask-text-intelligence) | [repo](https://github.com/deepgram-starters/flask-voice-agent) | [repo](https://github.com/deepgram-starters/flask-flux) | [repo](https://github.com/deepgram-starters/flask-flux-tts) | +| **django** | [repo](https://github.com/deepgram-starters/django-transcription) | [repo](https://github.com/deepgram-starters/django-live-transcription) | [repo](https://github.com/deepgram-starters/django-text-to-speech) | [repo](https://github.com/deepgram-starters/django-live-text-to-speech) | [repo](https://github.com/deepgram-starters/django-text-intelligence) | [repo](https://github.com/deepgram-starters/django-voice-agent) | [repo](https://github.com/deepgram-starters/django-flux) | [repo](https://github.com/deepgram-starters/django-flux-tts) | +| **go** | [repo](https://github.com/deepgram-starters/go-transcription) | [repo](https://github.com/deepgram-starters/go-live-transcription) | [repo](https://github.com/deepgram-starters/go-text-to-speech) | [repo](https://github.com/deepgram-starters/go-live-text-to-speech) | [repo](https://github.com/deepgram-starters/go-text-intelligence) | [repo](https://github.com/deepgram-starters/go-voice-agent) | [repo](https://github.com/deepgram-starters/go-flux) | — | +| **java** | [repo](https://github.com/deepgram-starters/java-transcription) | [repo](https://github.com/deepgram-starters/java-live-transcription) | [repo](https://github.com/deepgram-starters/java-text-to-speech) | [repo](https://github.com/deepgram-starters/java-live-text-to-speech) | [repo](https://github.com/deepgram-starters/java-text-intelligence) | [repo](https://github.com/deepgram-starters/java-voice-agent) | [repo](https://github.com/deepgram-starters/java-flux) | [repo](https://github.com/deepgram-starters/java-flux-tts) | +| **csharp** | [repo](https://github.com/deepgram-starters/csharp-transcription) | [repo](https://github.com/deepgram-starters/csharp-live-transcription) | [repo](https://github.com/deepgram-starters/csharp-text-to-speech) | [repo](https://github.com/deepgram-starters/csharp-live-text-to-speech) | [repo](https://github.com/deepgram-starters/csharp-text-intelligence) | [repo](https://github.com/deepgram-starters/csharp-voice-agent) | [repo](https://github.com/deepgram-starters/csharp-flux) | — | +| **rust** | [repo](https://github.com/deepgram-starters/rust-transcription) | [repo](https://github.com/deepgram-starters/rust-live-transcription) | [repo](https://github.com/deepgram-starters/rust-text-to-speech) | [repo](https://github.com/deepgram-starters/rust-live-text-to-speech) | [repo](https://github.com/deepgram-starters/rust-text-intelligence) | [repo](https://github.com/deepgram-starters/rust-voice-agent) | [repo](https://github.com/deepgram-starters/rust-flux) | — | +| **ruby** | [repo](https://github.com/deepgram-starters/ruby-transcription) | [repo](https://github.com/deepgram-starters/ruby-live-transcription) | [repo](https://github.com/deepgram-starters/ruby-text-to-speech) | [repo](https://github.com/deepgram-starters/ruby-live-text-to-speech) | [repo](https://github.com/deepgram-starters/ruby-text-intelligence) | [repo](https://github.com/deepgram-starters/ruby-voice-agent) | [repo](https://github.com/deepgram-starters/ruby-flux) | — | +| **php** | [repo](https://github.com/deepgram-starters/php-transcription) | [repo](https://github.com/deepgram-starters/php-live-transcription) | [repo](https://github.com/deepgram-starters/php-text-to-speech) | [repo](https://github.com/deepgram-starters/php-live-text-to-speech) | [repo](https://github.com/deepgram-starters/php-text-intelligence) | [repo](https://github.com/deepgram-starters/php-voice-agent) | [repo](https://github.com/deepgram-starters/php-flux) | — | +| **cpp** | [repo](https://github.com/deepgram-starters/cpp-transcription) | [repo](https://github.com/deepgram-starters/cpp-live-transcription) | [repo](https://github.com/deepgram-starters/cpp-text-to-speech) | [repo](https://github.com/deepgram-starters/cpp-live-text-to-speech) | [repo](https://github.com/deepgram-starters/cpp-text-intelligence) | [repo](https://github.com/deepgram-starters/cpp-voice-agent) | [repo](https://github.com/deepgram-starters/cpp-flux) | — | + +## Need something more specific? + +- **Focused feature snippets** (one feature, one language, < 50 lines) → `recipes` skill → +- **Third-party integrations** (Twilio, LiveKit, LangChain, Vercel AI SDK, Discord, etc.) → `examples` skill → +- **SDK-specific code skills** (idiomatic imports, async patterns, gotchas) → `npx skills add deepgram/deepgram-{lang}-sdk` — see the `api` skill for the 6 SDKs whose skills are publicly installable. + +## Related Deepgram skills + +- `api`: consolidated REST + WebSocket API reference +- `recipes`: minimal runnable feature snippets per language +- `examples`: full integration examples with third-party platforms +- `docs`: documentation finder +- `cli`: `deepctl`, including `dg init` for scaffolding a template from the terminal +- `setup-mcp`: Deepgram MCP server installation diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/docs.md b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/docs.md new file mode 100644 index 00000000..f9755fe0 --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/docs.md @@ -0,0 +1,106 @@ +--- +name: docs +description: > + Find the right Deepgram documentation for any task. Use whenever someone needs help locating + docs, understanding which API to use, or wants to ask questions about Deepgram. Covers all + product areas: speech-to-text (Nova, Flux STT), text-to-speech (Aura, Flux TTS), voice agents, + audio intelligence, and self-hosted deployments. +--- + +# Deepgram Documentation + +Find the right docs for what you're building with Deepgram. + +## Ask AI + +Have a question? Get answers from Deepgram's AI assistant at . + +## Documentation by Topic + +### Speech-to-Text (STT) + +Transcribe audio and video into text. Deepgram ships two actively maintained, next-gen model families — pick the one that matches your use case. + +- **Nova** (`/v1/listen`) — general-purpose transcription (captions, subtitles, batch files, live streams). Rich feature set including intelligence overlays (diarize, summarize, sentiment, topics, intents). +- **Flux STT** (`/v2/listen`) — conversational-audio transcription for voice agents and interactive assistants. Built-in turn-taking (EOT events, mid-session reconfig). + +Docs: +- [STT Getting Started (Nova)](https://developers.deepgram.com/docs/stt/getting-started) +- [Flux STT Quickstart](https://developers.deepgram.com/docs/flux/quickstart) +- [Nova 3 → Flux STT migration](https://developers.deepgram.com/docs/flux/nova-3-migration) +- [Flux STT language prompting](https://developers.deepgram.com/docs/flux/language-prompting) + +### Text-to-Speech (TTS) + +Convert text into natural-sounding speech. Deepgram ships two TTS model families on separate endpoints — the voices do not overlap. + +- **Aura** (`/v1/speak`) — the broadest voice catalog (English, Spanish, German, Dutch, French, Italian, Japanese) and compressed/containerized output. Use for one-shot synthesis and any non-English voice. +- **Flux TTS** (`/v2/speak`) — streaming-first, voice-agent-first synthesis. Turn-based lifecycle, barge-in with spoken-text feedback, and prosody that carries across turns. English at launch. + +Docs: +- [Text-to-Speech Docs (Aura)](https://developers.deepgram.com/docs/tts-rest) +- [Aura voices and languages](https://developers.deepgram.com/docs/tts-models) +- [Flux TTS Overview](https://developers.deepgram.com/docs/flux-tts/overview) +- [Flux TTS Streaming Quickstart](https://developers.deepgram.com/docs/flux-tts/quickstart) +- [Flux TTS Batch (REST) Quickstart](https://developers.deepgram.com/docs/flux-tts/batch) +- [Flux TTS batch vs streaming](https://developers.deepgram.com/docs/flux-tts/batch-vs-streaming) +- [Flux TTS voices](https://developers.deepgram.com/docs/flux-tts/voices) +- [Aura → Flux TTS migration](https://developers.deepgram.com/docs/flux-tts/migrating) + +### Voice Agent + +Build conversational voice agents powered by Deepgram. + +- [Voice Agent Docs](https://developers.deepgram.com/docs/voice-agent) +- [Voice Agent TTS models (Aura vs Flux TTS)](https://developers.deepgram.com/docs/voice-agent-tts-models) +- [Build a Flux TTS voice agent](https://developers.deepgram.com/docs/flux-tts/voice-agent) + +### Text and Audio Intelligence + +Analyze text and audio for sentiment, topics, intents, summaries, and more. + +- [Audio Intelligence Docs](https://developers.deepgram.com/docs/audio-intelligence) + +### Self-Hosted Deployments + +Run Deepgram on your own infrastructure. + +- [Self-Hosted Introduction](https://developers.deepgram.com/docs/self-hosted-introduction) + +### API Reference + +Full reference for all Deepgram REST and WebSocket APIs. + +- [API Reference](https://developers.deepgram.com/reference/deepgram-api-overview) + +## SDK-Specific Skills + +For language-idiomatic code patterns (imports, async idioms, error handling, type shapes), install the Deepgram SDK's own skills. Every Deepgram SDK publishes 7 product skills: + +```bash +npx skills add deepgram/deepgram-python-sdk # Python +npx skills add deepgram/deepgram-js-sdk # JavaScript / TypeScript +npx skills add deepgram/deepgram-java-sdk # Java +npx skills add deepgram/deepgram-go-sdk # Go +npx skills add deepgram/deepgram-rust-sdk # Rust +npx skills add deepgram/deepgram-dotnet-sdk # C# / .NET +``` + +Swift and Kotlin SDK skills are not listed because those repositories are not public and `npx skills add` cannot reach them. For browser work, open the `browser-agent` skill: it covers the four Browser Agent SDK packages published on npm (`@deepgram/agents`, `@deepgram/react`, `@deepgram/ui`, `@deepgram/agents-widget`). + +## Related Deepgram skills + +- `api`: consolidated REST + WebSocket API reference +- `recipes`: minimal runnable feature snippets per language +- `examples`: full integration examples with third-party platforms +- `starters`: runnable starter apps (framework × feature) +- `audio-intelligence`: the `/v1/listen` analysis parameters +- `text-intelligence`: `POST /v1/read` for text you already have +- `browser-agent`: running a voice agent in a browser +- `cli`: `deepctl` for shell and CI work +- `self-hosted`: running Deepgram on your own GPUs +- `setup-mcp`: Deepgram MCP server installation + +## MCP Server + +For direct documentation querying from your AI coding tool, use the `setup-mcp` skill to install the Deepgram MCP server. diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/regen_allowlist.py b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/regen_allowlist.py new file mode 100644 index 00000000..0fb83e3b --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/regen_allowlist.py @@ -0,0 +1,106 @@ +"""Regenerate allowlist.tsv: every deepgram/skills SKILL.md blob deepctl 0.2.16-0.3.2 could write. + +Not collected by pytest (no test_ prefix). Run it by hand before release: + python regen_allowlist.py > allowlist.tsv +Needs `gh` (the activity API gives push times and exposes any force-push) and +network access to pypi.org (release upload times). 0.2.16-0.3.2 downloaded +skills/{api,docs,setup-mcp,starters}/SKILL.md from main at install time and +fell back to its own cached copy (~/.deepctl/skills/repo_cache/.md) when +a download failed; <=0.2.15 fetched skills/mcp, so its cache never holds these. +A blob is listed when main served it after the first writer release was +published. live_releases names the releases that were current on PyPI while main +served it (any 0.2.16 to 0.3.2 install could fetch it). +""" + +import hashlib +import json +import subprocess +import sys +import urllib.request + +NAMES = ("api", "docs", "setup-mcp", "starters") +WRITERS = ("0.2.16", "0.2.17", "0.2.18", "0.2.19", "0.2.20", "0.2.21", "0.2.22") +WRITERS += ("0.2.23", "0.2.24", "0.2.25", "0.2.26", "0.3.0", "0.3.1", "0.3.2") +# v0.2.27 was tagged but never reached PyPI (twine rejected its metadata); its +# generator is byte-identical to 0.3.x, so it adds no blob and no window. + + +def run(*cmd: str) -> bytes: + return subprocess.run(cmd, capture_output=True, check=True).stdout + + +def main(repo: str) -> None: + acts = json.loads( + run( + "gh", + "api", + "--paginate", + "--slurp", + "repos/deepgram/skills/activity?ref=refs/heads/main&per_page=100", + ) + ) + acts = sorted((a for page in acts for a in page), key=lambda a: a["timestamp"]) + bad = [a for a in acts if a["activity_type"] == "force_push"] + assert not bad, f"force-push on main: {bad}" + pushed = {a["after"]: a["timestamp"] for a in acts} + history = run( + "git", "-C", repo, "rev-list", "--first-parent", "--reverse", "main" + ).split() + stray = set(pushed) - {c.decode() for c in history} + assert not stray, f"pushed tips missing from the clone's history: {stray}" + # Only pushed tips were ever served; a multi-commit push's inner commits never were. + history = [c for c in history if c.decode() in pushed] + pypi = json.load(urllib.request.urlopen("https://pypi.org/pypi/deepctl/json"))[ + "releases" + ] + released = {v: min(f["upload_time_iso_8601"] for f in pypi[v]) for v in WRITERS} + first_writer = released[WRITERS[0]] + head = history[-1].decode() + windows: dict[ + tuple[str, str], list[str] + ] = {} # (name, sha) -> [bytes, commit, start, end] + current: dict[str, tuple[str, str] | None] = dict.fromkeys(NAMES) + for c in map(bytes.decode, history): + at = pushed[c] + for n in NAMES: + r = subprocess.run( + ["git", "-C", repo, "show", f"{c}:skills/{n}/SKILL.md"], + capture_output=True, + ) + key = ( + (n, hashlib.sha256(r.stdout).hexdigest()) if r.returncode == 0 else None + ) + if current[n] and current[n] != key: + windows[current[n]][3] = at # Replaced (or deleted) at this push. + if key and current[n] != key: + windows.setdefault(key, [str(len(r.stdout)), c[:12], at, "9999"]) + windows[key][3] = "9999" # Served again from here. + current[n] = key + print( + f"# deepgram/skills main at {head[:12]}; writer releases 0.2.16-0.3.2 from PyPI;" + " live_releases: releases that were current on PyPI while main served it" + " (any 0.2.16 to 0.3.2 install could fetch it)" + ) + print("skill\tbytes\tsha256\tfirst_commit\tpushed_at\treplaced_at\tlive_releases") + for (n, sha), (size, c, start, end) in sorted( + windows.items(), key=lambda kv: (NAMES.index(kv[0][0]), kv[1][2]) + ): + if end <= first_writer: + print( + f"# dropped {n} {size}:{sha}: replaced {end}, before 0.2.16 ({first_writer})", + file=sys.stderr, + ) + continue + live = [ + v + for v in WRITERS + if released[v] < end + and (v == WRITERS[-1] or released[WRITERS[WRITERS.index(v) + 1]] > start) + ] + print( + f"{n}\t{size}\t{sha}\t{c}\t{start}\t{'' if end == '9999' else end}\t{','.join(live)}" + ) + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/setup-mcp.md b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/setup-mcp.md new file mode 100644 index 00000000..da2eb96c --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/setup-mcp.md @@ -0,0 +1,341 @@ +--- +name: setup-mcp +description: > + Set up a Deepgram MCP server for your AI coding tool. Offers three paths: the Deepgram CLI + MCP proxy (dg mcp), the standalone deepgram-mcp package, and the credential-free hosted + documentation MCP. Use whenever someone wants to install Deepgram's agentic tools, set up + the MCP server, or connect their editor to Deepgram. +--- + +# Install a Deepgram MCP Server + +You are setting up Deepgram MCP integration for the user. Follow these steps. + +## Step 1: Pick a path + +Three paths exist. Pick by whether the user has, or wants, a Deepgram API key. + +| Path | Server | Credentials | Install footprint | +|---|---|---|---| +| **A** | Deepgram CLI MCP proxy (`dg mcp`) | Deepgram API key **required** | Full CLI (`deepctl`) | +| **B** | Standalone `deepgram-mcp` | Deepgram API key **required** | One Python package | +| **C** | Hosted docs MCP (`/_mcp/server`) | **None** | Nothing to install | + +Decision rule: + +- The user already has the CLI, or wants `dg listen` / `dg speak` / `dg init` too → **Path A**. +- The user has an API key but wants only the MCP server, no CLI → **Path B**. +- The user has no API key, or wants something working in one command → **Path C**. + +A key-authenticated hosted variant of Paths A/B also exists at `api.dx.deepgram.com/kapa/mcp`, +with nothing to install — see "The kapa endpoints are not credential-free" below. + +Paths A and B are the same server: `dg mcp` wraps the `deepgram-mcp` package. Both proxy +Deepgram's developer API and fetch their tool list from Deepgram at runtime, so new tools +appear on reconnect without a package upgrade. As of this writing that list is a single +documentation and knowledge-source search tool (`search_deepgram_knowledge_sources`) — check +`tools/list` in the user's client for what is live rather than promising a tool set. + +Paths A/B and Path C both answer Deepgram questions from documentation, so installing more +than one is usually redundant. Path C is the only one that works with no credentials. + +## Step 2: Detect the environment + +Determine which AI coding tool the user is running. Check for: + +- **Claude Code** — look for a `.claude/` directory in the project or user home +- **Cursor** — look for a `.cursor/` directory in the project root +- **Windsurf** — look for a `.windsurf/` directory in the project root + +If multiple are detected, or none are detected, ask the user which tool they want to configure. + +## Step 3: Ask about scope + +Ask the user whether they want the MCP server configured: + +- **For this project only** (recommended for team repos) +- **Globally** (available in all projects) + +--- + +## Path A — Deepgram CLI MCP proxy (`dg mcp`) + +### A1. Install the CLI + +Check first: `dg --version` (or `deepctl --version`, or `where dg` on Windows). The package is +`deepctl` and installs three interchangeable binaries — `dg`, `deepctl`, and `deepgram`. + +```sh +# macOS / Linux — Homebrew (also brings in ffmpeg and portaudio) +brew install deepgram/tap/deepgram + +# macOS / Linux — install script +curl -fsSL https://deepgram.com/install.sh | sh + +# pip / uv / pipx +pip install deepctl +uv tool install deepctl +pipx install deepctl +``` + +```powershell +# Windows — PowerShell +iwr https://deepgram.com/install.ps1 -useb | iex +``` + +To upgrade, use the installer that put it there: `pip install -U deepctl`, +`uv tool upgrade deepctl`, `pipx upgrade deepctl`, `brew upgrade deepgram`, or re-run the install +script. `dg update --check-only` reports whether a newer release exists; on a pip install, bare +`dg update` reports `installation_method: null` instead of upgrading. + +The fully qualified Homebrew name matters: Homebrew 6 loads a third-party formula only after it +is trusted, and `brew install deepgram/tap/deepgram` trusts that one formula, where +`brew tap deepgram/tap && brew install deepgram` fails until a separate `brew trust` step. The +tap formula pins `deepctl-0.2.26`; pip, uv, and pipx install 0.3.1. + +### A2. Authenticate — required + +`dg mcp` will not start without credentials. Do this before configuring any editor: + +```sh +dg login # interactive; or dg login --api-key +dg whoami # confirm: "authenticated": true +``` + +`DEEPGRAM_API_KEY` in the environment works instead of `dg login`. Get a key at +. + +### A3. Configure the editor + +#### Claude Code + +```sh +# Project scope +claude mcp add deepgram --scope project dg mcp + +# User/global scope +claude mcp add deepgram dg mcp +``` + +#### Cursor + +Write or merge into the project's `.cursor/mcp.json`: + +```json +{ + "mcpServers": { + "deepgram": { + "type": "stdio", + "command": "dg", + "args": ["mcp"] + } + } +} +``` + +#### Windsurf + +Write or merge into the project's `.windsurf/mcp.json`, using the same object as Cursor above. + +#### Without a permanent install + +`uvx` and `pipx run` fetch `deepctl` on demand. Credentials still come from `dg login` or +`DEEPGRAM_API_KEY`: + +```json +{ + "mcpServers": { + "deepgram": { + "command": "uvx", + "args": ["deepctl", "mcp"] + } + } +} +``` + +#### Other tools + +- **Transport:** stdio +- **Command:** `dg` +- **Args:** `["mcp"]` + +`dg mcp --transport sse --port 8000` serves SSE instead, for clients that need HTTP. + +--- + +## Path B — Standalone `deepgram-mcp` + +The MCP server without the rest of the CLI. One package, one binary. + +```sh +pip install deepgram-mcp +export DEEPGRAM_API_KEY=your_key_here +``` + +`deepgram-mcp` is a PyPI package. The npm package of the same name is unrelated third-party code +that also asks for `DEEPGRAM_API_KEY`, so do not run `npx deepgram-mcp`. + +#### Claude Code + +```sh +claude mcp add deepgram -- deepgram-mcp +``` + +#### Cursor / Windsurf + +Write or merge into `.cursor/mcp.json` or the Windsurf MCP config: + +```json +{ + "mcpServers": { + "deepgram": { + "command": "deepgram-mcp", + "env": { + "DEEPGRAM_API_KEY": "your_key_here" + } + } + } +} +``` + +`--api-key` overrides the environment variable, and `--transport sse --port 8000` serves SSE. +Source: . + +--- + +## Path C — Hosted documentation MCP (no credentials) + +Use `https://developers.deepgram.com/_mcp/server`. It answers unauthenticated, needs no API +key, and exposes one tool, `searchDocs`, which returns documentation passages with source URLs. + +It is not a plain liveness URL. `HEAD` returns 404, a `GET` with the MCP +`Accept: application/json, text/event-stream` header returns 405, and a bare `GET` returns a +JSON descriptor of the server rather than an MCP response. Only a `POST` `initialize` exercises +the server; it answers 200 with `serverInfo.name` `fern-docs-mcp-server`: + +```sh +curl -s -X POST https://developers.deepgram.com/_mcp/server \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-03-26","capabilities":{},"clientInfo":{"name":"probe","version":"0"}}}' +``` + +#### Claude Code + +```sh +# Project scope +claude mcp add deepgram-docs --scope project --transport http https://developers.deepgram.com/_mcp/server + +# User/global scope +claude mcp add deepgram-docs --transport http https://developers.deepgram.com/_mcp/server +``` + +#### Cursor + +Write or merge into the project's `.cursor/mcp.json`: + +```json +{ + "mcpServers": { + "deepgram-docs": { + "type": "http", + "url": "https://developers.deepgram.com/_mcp/server" + } + } +} +``` + +#### Windsurf + +Write or merge into the project's `.windsurf/mcp.json`, using the same object as Cursor above. + +#### Other tools + +- **Type:** HTTP +- **URL:** `https://developers.deepgram.com/_mcp/server` + +### The kapa endpoints are not credential-free + +`https://api.dx.deepgram.com/kapa/mcp` and `https://deepgram.mcp.kapa.ai` both exist, and both +reject an unauthenticated request with HTTP 401 plus a `WWW-Authenticate: Bearer +resource_metadata=...` header, so a client that implements MCP's OAuth flow can connect to either. +They differ in whether a Deepgram API key works: + +- **`api.dx.deepgram.com/kapa/mcp` accepts a Deepgram API key.** Send it as either + `Authorization: Token ` or `Authorization: Bearer ` and `initialize` returns 200 from + `deepgram-mcp-relay`; an invalid key gets 401. `tools/list` returns the same single + `search_deepgram_knowledge_sources` tool as Paths A and B, so this is the hosted HTTP form of + the same server — useful when the user has a key but cannot install anything. Pass the key as a + header, or the client falls back to OAuth: + + ```sh + claude mcp add deepgram-relay --transport http https://api.dx.deepgram.com/kapa/mcp \ + --header "Authorization: Token $DEEPGRAM_API_KEY" + ``` +- **`deepgram.mcp.kapa.ai` does not.** A Deepgram API key gets 401 with either scheme. OAuth is + the only way in. + +Neither is the zero-setup option — use `/_mcp/server` for that. + +--- + +## Step 4: Confirm + +- **Claude Code** — run `/reload-plugins` to activate immediately, no restart needed. +- **Cursor / Windsurf / Other** — the user may need to restart or reload their tool. + +Then tell the user the server is configured, and check what it actually exposes before +describing it — have the client list its tools rather than naming tools from memory. + +For Path C, add: + +> Your tool can now search Deepgram's documentation directly — try asking about API +> parameters, voice agents, or model capabilities. + +Link them to [Deepgram Agentic Tools](https://developers.deepgram.com/developer-tools/agentic-tools) +for more details. Its two kapa URLs, `https://api.dx.deepgram.com/kapa/mcp` and +`https://deepgram.mcp.kapa.ai`, require credentials: an unauthenticated `initialize` returns 401. +The Docs MCP server at `https://developers.deepgram.com/_mcp/server` is the credential-free path. + +## Troubleshooting + +**`Error: DEEPGRAM_API_KEY is not set in the configuration file (...config.yaml) or environment variable.`** +followed by `Run deepctl login to configure the CLI with your Deepgram account.` +→ Path A with no credentials. `dg mcp` exits 1 before serving anything. Run `dg login`, or set +`DEEPGRAM_API_KEY`. Confirm with `dg whoami`. + +**`Error: No API key. Set DEEPGRAM_API_KEY or use --api-key.`** +→ Path B with no credentials. Export `DEEPGRAM_API_KEY`, put it in the server's `env` block, or +pass `--api-key`. + +**`! Needs authentication` in `claude mcp list`, or HTTP 401 `{"status_code":401,"detail":"Authentication required"}` / `{"error":"invalid_token"}`** +→ You are pointed at a kapa endpoint with no credentials. Switch to +`https://developers.deepgram.com/_mcp/server`, which needs none. To stay on +`api.dx.deepgram.com/kapa/mcp`, add `--header "Authorization: Token $DEEPGRAM_API_KEY"` — that +endpoint accepts a Deepgram API key. On `deepgram.mcp.kapa.ai` an API key does not work; let the +client run its OAuth flow instead. + +**`Server "deepgram-docs" is defined in multiple scopes with different endpoints`** +→ An earlier setup registered `deepgram-docs` at a kapa URL in user scope, and this one added a +different URL in project scope. OAuth tokens are stored per endpoint, so authenticating one does +not carry over. Keep one: `claude mcp remove deepgram-docs -s user` (or `-s project`). Check for +a pre-existing entry with `claude mcp get deepgram-docs` before adding, and pick a distinct +server name if the user wants to keep both. + +**`ImportError` mentioning `streamablehttp_client` on startup** +→ An incompatible `mcp` package. `deepgram-mcp` imports `streamablehttp_client` from +`mcp.client.streamable_http`, which `mcp` 2.0 removed. Install into a clean environment, or pin +`mcp>=1.0.0,<2.0.0`. Installing `deepctl` pins this for you. + +**The server connects but exposes fewer tools than expected** +→ Expected. Paths A and B fetch their tool list from Deepgram at runtime, so it reflects what +the API serves right now, not what the package version implies. Reconnect to pick up new tools. + +**Anything else on Path A** +→ Verify `dg --version` works and `dg mcp` runs in a terminal without errors, then +`dg update --check-only` to see whether a newer release exists. + +## Sources + +- Deepgram CLI: +- `deepgram-mcp`: +- Deepgram Agentic Tools: diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/starters.md b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/starters.md new file mode 100644 index 00000000..75bf37e5 --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/starters.md @@ -0,0 +1,205 @@ +--- +name: starters +description: > + Clone a ready-to-run Deepgram demo app and start building on top of it. Use whenever someone + wants a quick working demo, needs to prototype with Deepgram, or is starting a new project + that uses speech-to-text, text-to-speech, voice agents, audio intelligence, or live streaming. + Match the user's language, framework, and desired Deepgram feature to the right starter. +--- + +# Deepgram Starter Apps + +Clone a working demo and start building. Every starter is a minimal, runnable app you can extend. + +## 1. Pick Your Feature + +What do you want to build? + +- **Transcribe a file** → `transcription` — send audio/video, get text back (REST, Nova) +- **Transcribe a live stream** → `live-transcription` — real-time speech-to-text (WebSocket, Nova) +- **Generate speech** → `text-to-speech` — send text, get audio back (REST, Aura) +- **Stream speech** → `live-text-to-speech` — real-time text-to-audio (WebSocket, Aura) +- **Analyze text** → `text-intelligence` — sentiment, topics, intents, summaries over text you + already have (REST, `/v1/read`) +- **Build a voice agent** → `voice-agent` — conversational AI agent (WebSocket, agent.deepgram.com) +- **Conversational STT with turn detection** → `flux` — Deepgram Flux STT for voice agents and interactive assistants (WebSocket, `/v2/listen`) +- **Turn-based TTS for a voice agent** → `flux-tts` — Deepgram Flux TTS, streaming synthesis with barge-in (WebSocket, `/v2/speak`) + +**There is no audio-intelligence starter.** `text-intelligence` is text-only — it posts text you +already have to `/v1/read`. No `{framework}-audio-intelligence` repository exists in +`deepgram-starters` for any framework, so don't construct those URLs. To run intelligence features +(summarization, sentiment, topics, intents) over *audio*, they are query parameters on +`/v1/listen`, not a separate starter: clone the `transcription` starter for your framework and add +the parameters to its existing request. See the `api` skill for which features `/v1/listen` +supports. + +**Nova vs Flux STT for speech-to-text:** use `transcription` or `live-transcription` (Nova, `/v1/listen`) for general-purpose transcription, captions, and batch workloads. Use `flux` (Flux STT, `/v2/listen`) when you need built-in turn detection for conversational audio. See the `api` skill for a full comparison. + +**Aura vs Flux TTS for text-to-speech:** use `text-to-speech` or `live-text-to-speech` (Aura, `/v1/speak`) for one-shot synthesis, non-English voices, and compressed audio. Use `flux-tts` (Flux TTS, `/v2/speak`) when you're streaming LLM output to a speaker and need a turn lifecycle and barge-in. See the `api` skill for a full comparison. + +**Flux TTS starters exist for `node`, `flask`, `fastapi`, `django`, and `java` only** — these are the five apps Deepgram officially publishes at [Flux TTS template apps](https://developers.deepgram.com/docs/flux-tts/template-apps). There is no `flux-tts` starter for the other frameworks; don't construct those URLs. For an unsupported framework, start from the `api` skill's Flux TTS section and the SDK skills instead. + +## 2. Pick Your Stack + +| Language | Frameworks | +|----------|------------| +| JavaScript | `node` | +| TypeScript | `bun`, `deno` | +| Python | `fastapi`, `flask`, `django` | +| Go | `go` | +| Java | `java` | +| C# | `csharp` | +| Rust | `rust` | +| Ruby | `ruby` | +| PHP | `php` | +| C++ | `cpp` | + +## 3. Clone and Run + +Every starter lives at `https://github.com/deepgram-starters/{framework}-{feature}` — framework +first, feature second. Clone **with submodules**; each starter vendors two git submodules — its +browser frontend at `frontend/` and the shared starter contracts at `contracts/` — and a plain +`git clone` leaves both directories empty and the app unrunnable: + +```sh +git clone --recurse-submodules https://github.com/deepgram-starters/{framework}-{feature}.git +cd {framework}-{feature} +``` + +In 80 of the 96 starters, both submodule URLs in `.gitmodules` are SSH (`git@github.com:...`) +even though both repositories are public, so `--recurse-submodules` fails with +`Host key verification failed` unless the user has a GitHub SSH key. The other 16 use HTTPS URLs +and clone without a key: 12 of the 13 `{framework}-live-transcription` starters (every one except +`rust-live-transcription`) plus `csharp-voice-agent`, `django-voice-agent`, `flask-voice-agent`, +and `node-voice-agent`. Without an SSH key, rewrite SSH to HTTPS for the clone. The rewrite +changes nothing on the 16 HTTPS starters, so it is safe to use on every starter: + +```sh +git -c url."https://github.com/".insteadOf="git@github.com:" \ + clone --recurse-submodules https://github.com/deepgram-starters/{framework}-{feature}.git +``` + +The starter's own `make init` runs `git submodule update --init --recursive` and installs +dependencies, but it inherits the URLs in `.gitmodules`. On the 80 SSH starters it fails +identically without a key, so it is the path for users who **have** SSH set up (or for one of the +16 HTTPS starters), not a workaround for users who don't. + +Set your API key and follow the README: + +```sh +export DEEPGRAM_API_KEY=your_key_here +``` + +Get an API key at . + +### Or scaffold with the CLI + +The [Deepgram CLI](https://github.com/deepgram/cli) has a scaffolder that finds and clones a +starter for you: + +```sh +dg init --list # browse templates +dg init --list --search python # filter +dg init node-transcription # clone into ./node-transcription +dg init node-transcription --dir ./my-app +``` + +**`dg init` does not solve the submodule problem.** It runs a plain clone, so `frontend/` and +`contracts/` land empty, and it still prints `Done! … is ready` and `"status": "success"`. Adding +`--install` runs the starter's `make check-prereqs && make init`, which hits the same `.gitmodules` +URLs: on the 80 SSH starters it fails with `Host key verification failed`, and `dg init` reports +success anyway. Without a GitHub SSH key, finish the checkout by hand after `dg init`: + +```sh +cd my-app +git -c url."https://github.com/".insteadOf="git@github.com:" \ + submodule update --init --recursive +``` + +`dg init` is also marked alpha, and its templates gallery is a separate list from the matrix +below rather than a subset of it. It carries 44 templates with no `flux` or `flux-tts` entries; +it still lists `sinatra-transcription`, whose repository is archived and private, so the clone +returns 404 for anyone outside Deepgram; and it lists `nextjs-*` templates that now redirect out +of `deepgram-starters` to `deepgram-devs`, which is why there is no `nextjs` row below. Treat +the matrix as authoritative and fall back to `git clone`. See the `cli` skill for installing +`deepctl` and for the rest of `dg init`. + +## The `{feature}-html` repos are not starters + +The `deepgram-starters` org also contains `transcription-html`, `live-transcription-html`, +`text-to-speech-html`, `live-text-to-speech-html`, `text-intelligence-html`, `voice-agent-html`, +`flux-html`, and `flux-tts-html`. **Do not clone these and do not offer them as starters.** Each +is the shared browser frontend that a backend starter pulls in as its `frontend/` submodule — +`node-transcription` vendors `transcription-html`, `flask-voice-agent` vendors `voice-agent-html`, +`node-flux-tts` and `java-flux-tts` both vendor `flux-tts-html`, and so on. Seven of the eight +say so in their own README ("This is a frontend submodule - do not use directly"); `flux-tts-html` +carries no such warning but is vendored the same way. None of them serve an API, so none of them +run standalone. Clone the backend starter instead and the right frontend arrives with it. + +They also invert the naming rule. The starter pattern is `{framework}-{feature}`, but these are +`{feature}-html` — and the mirror-image names do **not** exist, so do not construct them: +`deepgram-starters/html-transcription` is a 404. There is no vanilla-HTML row in the matrix +because there is no standalone browser starter; for browser-only work, clone the `node` starter +for the feature you want and read its `frontend/` directory. + +## Examples + +**"I want to build a voice agent in Python"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/fastapi-voice-agent.git` + +**"I need live transcription in my Node app"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/node-live-transcription.git` + +**"I want to add text-to-speech to my Go service"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/go-text-to-speech.git` + +**"I want to analyze audio for sentiment in C#"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/csharp-text-intelligence.git` + +**"I want streaming TTS with barge-in for my Node voice agent"** +→ `git clone --recurse-submodules https://github.com/deepgram-starters/node-flux-tts.git` + +**"I want a plain browser/HTML demo"** +→ There is no standalone HTML starter. Clone `node-{feature}` and work in its `frontend/` +directory — that is the same browser code the `{feature}-html` submodule holds. + +## All Starters + +Every URL below is a real, published, non-archived repository, and the table is the complete +set: 13 frameworks × 7 features, plus `flux-tts` for the five frameworks that have it. A cell +showing `—` means that starter does not exist; don't construct the URL. + +The `java-flux-tts` README clones with a plain `git clone`, without `--recurse-submodules`, while +its `.gitmodules` points both submodules at SSH URLs, so following its Maven steps leaves +`frontend/` and `contracts/` empty. Use the clone command in section 3 instead. + +| | transcription | live-transcription | text-to-speech | live-text-to-speech | text-intelligence | voice-agent | flux | flux-tts | +|---|---|---|---|---|---|---|---|---| +| **node** | [repo](https://github.com/deepgram-starters/node-transcription) | [repo](https://github.com/deepgram-starters/node-live-transcription) | [repo](https://github.com/deepgram-starters/node-text-to-speech) | [repo](https://github.com/deepgram-starters/node-live-text-to-speech) | [repo](https://github.com/deepgram-starters/node-text-intelligence) | [repo](https://github.com/deepgram-starters/node-voice-agent) | [repo](https://github.com/deepgram-starters/node-flux) | [repo](https://github.com/deepgram-starters/node-flux-tts) | +| **bun** | [repo](https://github.com/deepgram-starters/bun-transcription) | [repo](https://github.com/deepgram-starters/bun-live-transcription) | [repo](https://github.com/deepgram-starters/bun-text-to-speech) | [repo](https://github.com/deepgram-starters/bun-live-text-to-speech) | [repo](https://github.com/deepgram-starters/bun-text-intelligence) | [repo](https://github.com/deepgram-starters/bun-voice-agent) | [repo](https://github.com/deepgram-starters/bun-flux) | — | +| **deno** | [repo](https://github.com/deepgram-starters/deno-transcription) | [repo](https://github.com/deepgram-starters/deno-live-transcription) | [repo](https://github.com/deepgram-starters/deno-text-to-speech) | [repo](https://github.com/deepgram-starters/deno-live-text-to-speech) | [repo](https://github.com/deepgram-starters/deno-text-intelligence) | [repo](https://github.com/deepgram-starters/deno-voice-agent) | [repo](https://github.com/deepgram-starters/deno-flux) | — | +| **fastapi** | [repo](https://github.com/deepgram-starters/fastapi-transcription) | [repo](https://github.com/deepgram-starters/fastapi-live-transcription) | [repo](https://github.com/deepgram-starters/fastapi-text-to-speech) | [repo](https://github.com/deepgram-starters/fastapi-live-text-to-speech) | [repo](https://github.com/deepgram-starters/fastapi-text-intelligence) | [repo](https://github.com/deepgram-starters/fastapi-voice-agent) | [repo](https://github.com/deepgram-starters/fastapi-flux) | [repo](https://github.com/deepgram-starters/fastapi-flux-tts) | +| **flask** | [repo](https://github.com/deepgram-starters/flask-transcription) | [repo](https://github.com/deepgram-starters/flask-live-transcription) | [repo](https://github.com/deepgram-starters/flask-text-to-speech) | [repo](https://github.com/deepgram-starters/flask-live-text-to-speech) | [repo](https://github.com/deepgram-starters/flask-text-intelligence) | [repo](https://github.com/deepgram-starters/flask-voice-agent) | [repo](https://github.com/deepgram-starters/flask-flux) | [repo](https://github.com/deepgram-starters/flask-flux-tts) | +| **django** | [repo](https://github.com/deepgram-starters/django-transcription) | [repo](https://github.com/deepgram-starters/django-live-transcription) | [repo](https://github.com/deepgram-starters/django-text-to-speech) | [repo](https://github.com/deepgram-starters/django-live-text-to-speech) | [repo](https://github.com/deepgram-starters/django-text-intelligence) | [repo](https://github.com/deepgram-starters/django-voice-agent) | [repo](https://github.com/deepgram-starters/django-flux) | [repo](https://github.com/deepgram-starters/django-flux-tts) | +| **go** | [repo](https://github.com/deepgram-starters/go-transcription) | [repo](https://github.com/deepgram-starters/go-live-transcription) | [repo](https://github.com/deepgram-starters/go-text-to-speech) | [repo](https://github.com/deepgram-starters/go-live-text-to-speech) | [repo](https://github.com/deepgram-starters/go-text-intelligence) | [repo](https://github.com/deepgram-starters/go-voice-agent) | [repo](https://github.com/deepgram-starters/go-flux) | — | +| **java** | [repo](https://github.com/deepgram-starters/java-transcription) | [repo](https://github.com/deepgram-starters/java-live-transcription) | [repo](https://github.com/deepgram-starters/java-text-to-speech) | [repo](https://github.com/deepgram-starters/java-live-text-to-speech) | [repo](https://github.com/deepgram-starters/java-text-intelligence) | [repo](https://github.com/deepgram-starters/java-voice-agent) | [repo](https://github.com/deepgram-starters/java-flux) | [repo](https://github.com/deepgram-starters/java-flux-tts) | +| **csharp** | [repo](https://github.com/deepgram-starters/csharp-transcription) | [repo](https://github.com/deepgram-starters/csharp-live-transcription) | [repo](https://github.com/deepgram-starters/csharp-text-to-speech) | [repo](https://github.com/deepgram-starters/csharp-live-text-to-speech) | [repo](https://github.com/deepgram-starters/csharp-text-intelligence) | [repo](https://github.com/deepgram-starters/csharp-voice-agent) | [repo](https://github.com/deepgram-starters/csharp-flux) | — | +| **rust** | [repo](https://github.com/deepgram-starters/rust-transcription) | [repo](https://github.com/deepgram-starters/rust-live-transcription) | [repo](https://github.com/deepgram-starters/rust-text-to-speech) | [repo](https://github.com/deepgram-starters/rust-live-text-to-speech) | [repo](https://github.com/deepgram-starters/rust-text-intelligence) | [repo](https://github.com/deepgram-starters/rust-voice-agent) | [repo](https://github.com/deepgram-starters/rust-flux) | — | +| **ruby** | [repo](https://github.com/deepgram-starters/ruby-transcription) | [repo](https://github.com/deepgram-starters/ruby-live-transcription) | [repo](https://github.com/deepgram-starters/ruby-text-to-speech) | [repo](https://github.com/deepgram-starters/ruby-live-text-to-speech) | [repo](https://github.com/deepgram-starters/ruby-text-intelligence) | [repo](https://github.com/deepgram-starters/ruby-voice-agent) | [repo](https://github.com/deepgram-starters/ruby-flux) | — | +| **php** | [repo](https://github.com/deepgram-starters/php-transcription) | [repo](https://github.com/deepgram-starters/php-live-transcription) | [repo](https://github.com/deepgram-starters/php-text-to-speech) | [repo](https://github.com/deepgram-starters/php-live-text-to-speech) | [repo](https://github.com/deepgram-starters/php-text-intelligence) | [repo](https://github.com/deepgram-starters/php-voice-agent) | [repo](https://github.com/deepgram-starters/php-flux) | — | +| **cpp** | [repo](https://github.com/deepgram-starters/cpp-transcription) | [repo](https://github.com/deepgram-starters/cpp-live-transcription) | [repo](https://github.com/deepgram-starters/cpp-text-to-speech) | [repo](https://github.com/deepgram-starters/cpp-live-text-to-speech) | [repo](https://github.com/deepgram-starters/cpp-text-intelligence) | [repo](https://github.com/deepgram-starters/cpp-voice-agent) | [repo](https://github.com/deepgram-starters/cpp-flux) | — | + +## Need something more specific? + +- **Focused feature snippets** (one feature, one language, < 50 lines) → `recipes` skill → +- **Third-party integrations** (Twilio, LiveKit, LangChain, Vercel AI SDK, Discord, etc.) → `examples` skill → +- **SDK-specific code skills** (idiomatic imports, async patterns, gotchas) → `npx skills add deepgram/deepgram-{lang}-sdk` — see the `api` skill for the 6 SDKs whose skills are publicly installable. + +## Related Deepgram skills + +- `api`: consolidated REST + WebSocket API reference +- `recipes`: minimal runnable feature snippets per language +- `examples`: full integration examples with third-party platforms +- `docs`: documentation finder +- `cli`: `deepctl`, including `dg init` for scaffolding a template from the terminal +- `setup-mcp`: Deepgram MCP server installation diff --git a/packages/deepctl-core/tests/unit/fixtures/legacy_v03/v032_writer.py b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/v032_writer.py new file mode 100644 index 00000000..78ad6730 --- /dev/null +++ b/packages/deepctl-core/tests/unit/fixtures/legacy_v03/v032_writer.py @@ -0,0 +1,41 @@ +"""How deepctl v0.3.2 wrote its skill files, quoted so tests can regenerate them. + +Quoted from tag v0.3.2, packages/deepctl-core/src/deepctl_core/skill_generator.py: +- l.711-719 ``SkillGenerator._write_repo_skills`` (the base class, which Cursor, + Cline, Amazon Q and Aider inherit): the joined file. +- l.796-797 ``CodexGenerator._BEGIN`` / ``_END``. +- l.815-832 ``CodexGenerator._merge`` and ``_write_repo_skills``: the section in + Codex's file; Gemini and OpenCode copy them byte for byte. +Only ``self`` and the file I/O are replaced by parameters and return values. +The skills dict is ordered api, docs, setup-mcp, starters (``skill_names``, l.87). +v0.3.2 used ``path.read_text()``/``write_text()``: universal newlines on read, and +"\\n" written as-is on POSIX. +""" + +_BEGIN = "" +_END = "" + + +def joined(repo_skills: dict[str, str]) -> str: + # l.713 + combined = "\n\n---\n\n".join(repo_skills.values()) + return combined + + +def _merge(existing: str | None, section: str) -> str: + # l.815-825, with ``path.exists()`` / ``path.read_text()`` as ``existing``. + if existing is None: + return section + if _BEGIN in existing: + before = existing[: existing.index(_BEGIN)] + after_end = existing.find(_END) + after = existing[after_end + len(_END) :] if after_end != -1 else "" + return before + section + after.lstrip("\n") + return existing.rstrip("\n") + "\n\n" + section + + +def shared(repo_skills: dict[str, str], existing: str | None) -> str: + # l.827-832 + combined = "\n\n---\n\n".join(repo_skills.values()) + wrapped = f"{_BEGIN}\n{combined}\n{_END}\n" + return _merge(existing, wrapped) diff --git a/packages/deepctl-core/tests/unit/test_legacy_v03.py b/packages/deepctl-core/tests/unit/test_legacy_v03.py new file mode 100644 index 00000000..2b4dacd3 --- /dev/null +++ b/packages/deepctl-core/tests/unit/test_legacy_v03.py @@ -0,0 +1,1469 @@ +"""deepctl 0.3.x cleanup: remove only the files and sections deepctl can prove it wrote.""" + +import contextlib +import csv +import hashlib +import importlib.util +import json +import os +import shutil +import signal +import stat +from pathlib import Path + +import pytest +from deepctl_core import output, skill_bundle +from deepctl_core import skill_generator as sg +from deepctl_core.skill_bundle import RepoSkill + +POSIX = pytest.mark.skipif(os.name == "nt", reason="POSIX-only filesystem behavior") +pytestmark = pytest.mark.skipif( + os.name == "nt", reason="legacy cleanup is intentionally disabled on Windows" +) +REF = skill_bundle.DEFAULT_SKILLS_COMMIT +FIX = Path(__file__).parent / "fixtures" / "legacy_v03" +NAMES = ("api", "docs", "setup-mcp", "starters") +BLOB = {n: (FIX / f"{n}.md").read_bytes() for n in NAMES} +JOINED = (FIX / "deepctl.mdc").read_bytes() +BLOCK = (FIX / "GEMINI.md").read_bytes() +SHARED = { + "codex": ".codex/instructions.md", + "gemini": ".gemini/GEMINI.md", + "opencode": ".opencode/agents.md", +} +STANDALONE = {"cursor": ".cursor/rules/deepctl.mdc", "cline": ".cline/rules/deepctl.md"} +DIFFERS = "it differs from every deepgram/skills version deepctl 0.3.x copied" + + +def _writer(): + spec = importlib.util.spec_from_file_location("v032_writer", FIX / "v032_writer.py") + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +V032 = _writer() + + +@pytest.fixture(autouse=True) +def _throwaway_home(tmp_path, monkeypatch): + home = tmp_path / "home" + home.mkdir() + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.setenv("HOME", str(home)) + monkeypatch.setenv("USERPROFILE", str(home)) + monkeypatch.setattr(sg, "_SKILLS_DIR", home / ".deepctl" / "skills") + monkeypatch.setattr(sg, "_STATE_FILE", home / ".deepctl" / "skills" / "skills.json") + monkeypatch.setenv("COLUMNS", "400") + for con in (output.console, output.stderr_console): + monkeypatch.setattr(con, "_width", 400) + monkeypatch.delenv(skill_bundle.REF_ENV_VAR, raising=False) + return home + + +@pytest.fixture(autouse=True) +def _pinned_output(): + saved = dict(output._output_config) + output._output_config.update(agentic=True, format="default", quiet=False) + yield + output._output_config.clear() + output._output_config.update(saved) + + +def gen(cli): + return next(g for g in sg.get_all_generators() if g.cli_name == cli) + + +def at(rel): + return Path.home().joinpath(*rel.split("/")) + + +def claude(name): + return at(f".claude/commands/deepgram/{name}.md") + + +def bundle(tmp, names=NAMES): + skills = [] + for name in names: + folder = Path(tmp) / "bundle" / "skills" / name + folder.mkdir(parents=True, exist_ok=True) + (folder / "SKILL.md").write_bytes(f"---\nname: {name}\n---\nnew\n".encode()) + skills.append(RepoSkill(name, folder)) + return skills + + +def seed(files, record=True, extra=None): + """Write ``files`` ({path: bytes}) and 0.3.x's record listing them.""" + try: + state = disk() # Keep the folder records of an earlier install. + except FileNotFoundError: + state = {"installed_skills": {}, "auto_update": True} + for path, data in files.items(): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(data) + cli = next(c for c, rels in sg._V03_PATHS.items() if path in map(at, rels)) + if record: + rec = state["installed_skills"].setdefault(cli, {"paths": []}) + rec["paths"] = [*rec["paths"], str(path)] + state["installed_skills"].update(extra or {}) + sg._STATE_FILE.parent.mkdir(parents=True, exist_ok=True) + sg._STATE_FILE.write_text(json.dumps(state), encoding="utf-8") + + +def install(tmp, cli="claude", names=NAMES): + return sg.install_tool(gen(cli), bundle(tmp, names), ref=REF, version="0.0.0") + + +def disk(): + return json.loads(sg._STATE_FILE.read_bytes()) + + +def legacy_paths(cli): + return disk()["installed_skills"].get(cli, {}).get("paths") + + +def folder_paths(cli, names=NAMES): + return [str(gen(cli).skills_root() / n) for n in names] + + +def err(capsys): + return " ".join(capsys.readouterr().err.split()) + + +def note(key, **kw): + return " ".join(sg._msg(key, **kw).split()) + + +def ident(path): + st = os.lstat(path) + return st.st_ino, path.read_bytes() + + +def crlf(data): + return data.replace(b"\n", b"\r\n") + + +def link_kept(tmp_path, capsys, cli, path, data, dangling): + """A link at a legacy path, dangling or not, is kept and warned about once.""" + target = tmp_path / "target.md" + target.write_bytes(data) + seed({path: b""}) + path.unlink() + try: + path.symlink_to(os.path.relpath(target, path.parent)) + except (OSError, NotImplementedError) as exc: + pytest.skip(f"cannot create a symlink here: {exc}") + if dangling: + target.unlink() + install(tmp_path, cli) + assert path.is_symlink() + assert dangling or target.read_bytes() == data + assert "(it is a link)" in err(capsys) + install(tmp_path, cli) + assert err(capsys) == "" + + +class TestAllowlist: + def test_allowlist_matches_trace(self): + lines = (FIX / "allowlist.tsv").read_text(encoding="utf-8").splitlines() + assert "0fc13fa" in lines[0] + rows = list(csv.DictReader(lines[1:], delimiter="\t")) + assert len(rows) == 29 + assert all(r["live_releases"] for r in rows) + want = { + n: {f"{r['bytes']}:{r['sha256']}" for r in rows if r["skill"] == n} + for n in NAMES + } + assert {n: set(v) for n, v in sg._V03_BLOBS.items()} == want + assert sum(len(v) for v in sg._V03_BLOBS.values()) == 29 + + def test_fixtures_regenerate_from_v032_writer(self, tmp_path, capsys): + skills = {n: BLOB[n].decode("utf-8") for n in NAMES} + for n in NAMES: + entry = f"{len(BLOB[n])}:{hashlib.sha256(BLOB[n]).hexdigest()}" + assert entry in sg._V03_BLOBS[n] + joined = V032.joined(skills).encode("utf-8") + assert joined == JOINED + assert hashlib.sha256(joined).hexdigest().startswith("c6302c78") + seed({at(STANDALONE["cursor"]): joined}) + install(tmp_path, "cursor") + assert not at(STANDALONE["cursor"]).exists() + + +class TestStandalone: + def test_claude_files_removed_records_cleared(self, tmp_path, capsys): + seed({claude(n): BLOB[n] for n in NAMES}) + mine = claude("mine") + mine.write_bytes(b"the user's own command") + placed, _ = install(tmp_path) + assert len(placed) == 4 + assert [n for n in NAMES if claude(n).exists()] == [] + assert mine.read_bytes() == b"the user's own command" # The dir stays. + text = err(capsys) + assert "INFO: Removed deepctl 0.3.x files for Claude Code:" in text + assert str(claude("api")) in text + assert "WARN" not in text + state = disk() + assert "v03" not in state["skill_folders"]["claude"] + assert state["skill_folders"]["claude"]["skills_ref"] == REF + assert legacy_paths("claude") == folder_paths("claude", sorted(NAMES)) + + def test_v03_cleared_after_cleanup(self, tmp_path): + seed({claude(n): BLOB[n] for n in NAMES}) + state = disk() + state["skill_folders"] = { + "claude": {"folders": {}, "v03": True, "skills_ref": "old"} + } + sg._STATE_FILE.write_text(json.dumps(state), encoding="utf-8") + install(tmp_path) + state = disk() + assert "v03" not in state["skill_folders"]["claude"] + assert state["skill_folders"]["claude"]["skills_ref"] == REF + assert "claude" in state["installed_skills"] + assert legacy_paths("claude") == folder_paths("claude", sorted(NAMES)) + + def test_stale_v03_flag_popped_without_legacy_paths(self, tmp_path, capsys): + state = { + "installed_skills": {"claude": {"paths": folder_paths("claude")}}, + "skill_folders": {"claude": {"folders": {}, "v03": True}}, + } + sg._STATE_FILE.parent.mkdir(parents=True, exist_ok=True) + sg._STATE_FILE.write_text(json.dumps(state), encoding="utf-8") + install(tmp_path) + assert "v03" not in disk()["skill_folders"]["claude"] + assert err(capsys) == "" + + def test_command_dir_keeps_retained_legacy_copies(self, tmp_path): + seed({claude(n): BLOB[n] for n in NAMES}) + install(tmp_path) + kept = [ + n + for n in os.listdir(claude("api").parent) + if n.startswith(".deepctl-kept-v03-") + ] + assert len(kept) == len(NAMES) + + @pytest.mark.parametrize("cli", ["cursor", "cline"]) + @pytest.mark.parametrize( + "names", [NAMES, ("api", "starters"), ("docs",), ("api", "docs", "setup-mcp")] + ) + @pytest.mark.parametrize("eol", ["lf", "crlf"]) + def test_joined_file_removed(self, tmp_path, capsys, cli, names, eol): + data = V032.joined({n: BLOB[n].decode() for n in names}).encode() + data = crlf(data) if eol == "crlf" else data + seed({at(STANDALONE[cli]): data}) + install(tmp_path, cli) + assert not at(STANDALONE[cli]).exists() + assert at(STANDALONE[cli]).parent.is_dir() # Only Claude's folder is ours. + assert "Removed deepctl 0.3.x files for" in err(capsys) + assert legacy_paths(cli) == folder_paths(cli, sorted(NAMES)) + assert "v03" not in disk()["skill_folders"][cli] + + @pytest.mark.parametrize( + "data", + [ + BLOB["api"] + b"x", + BLOB["api"][:-2] + b"X\n", # Same size, one byte changed. + BLOB["api"].replace(b"\n", b"\r\n", 1), # Mixed line endings. + b"", + b"---\nname: api\n---\nmy own api notes\n", + BLOB["docs"], # Another skill's text at api.md. + BLOB["api"] + b"\n\n---\n\n", + ], + ids=["plus-byte", "same-size", "mixed-eol", "empty", "own", "docs", "sep"], + ) + def test_user_api_md_with_name_api_survives(self, tmp_path, capsys, data): + """B8: only exact deepgram/skills bytes prove; frontmatter never does.""" + api = claude("api") + seed({api: data}) + before = ident(api) + install(tmp_path) + assert ident(api) == before + text = err(capsys) + assert note("E33", path=api, why=DIFFERS) in text + assert text.count(str(api)) == 1 + assert str(api) not in legacy_paths("claude") + assert "v03" not in disk()["skill_folders"]["claude"] + install(tmp_path) # Warned once, then untracked: silent. + assert err(capsys) == "" + assert ident(api) == before + + def test_quiet_run_keeps_the_record_so_a_later_run_warns(self, tmp_path, capsys): + api = claude("api") + seed({api: b"mine"}) + output._output_config["quiet"] = True + install(tmp_path) + assert err(capsys) == "" + assert str(api) in legacy_paths("claude") + output._output_config["quiet"] = False + install(tmp_path) + assert note("E33", path=api, why=DIFFERS) in err(capsys) + install(tmp_path) + assert err(capsys) == "" + assert api.read_bytes() == b"mine" + + def test_joined_with_a_reordered_or_repeated_skill_kept(self, tmp_path, capsys): + sep = b"\n\n---\n\n" + for data in (BLOB["docs"] + sep + BLOB["api"], BLOB["api"] + sep + BLOB["api"]): + seed({at(STANDALONE["cursor"]): data}) + install(tmp_path, "cursor") + assert at(STANDALONE["cursor"]).read_bytes() == data + assert DIFFERS in err(capsys) + + def test_unrecorded_unprovable_file_is_silent(self, tmp_path, capsys): + api = claude("api") + seed({api: b"mine"}, record=False) + install(tmp_path) + assert api.read_bytes() == b"mine" + assert err(capsys) == "" + + def test_gate_keeps_files_whose_folder_did_not_land(self, tmp_path, capsys): + files = {claude(n): BLOB[n] for n in NAMES} + files[at(STANDALONE["cursor"])] = JOINED + seed(files) + without_api = ("docs", "setup-mcp", "starters") + install(tmp_path, "claude", without_api) + install(tmp_path, "cursor", without_api) + assert claude("api").read_bytes() == BLOB["api"] + assert not claude("docs").exists() + assert at(STANDALONE["cursor"]).read_bytes() == JOINED + assert "WARN" not in err(capsys) + assert legacy_paths("cursor") == [str(at(STANDALONE["cursor"]))] + assert str(claude("api")) in legacy_paths("claude") + assert disk()["skill_folders"]["claude"]["v03"] is True + + def test_amazonq_and_aider_files_untouched(self, tmp_path, capsys): + q, aider = ( + at(".amazonq/rules/deepctl.md"), + at(".deepctl/skills/deepctl-conventions.md"), + ) + for p in (q, aider): + p.parent.mkdir(parents=True, exist_ok=True) + p.write_bytes(JOINED) + extra = {"amazonq": {"paths": [str(q)]}, "aider": {"paths": [str(aider)]}} + seed({claude("api"): BLOB["api"]}, extra=extra) + for g in sg.get_all_generators(): + sg.install_tool(g, bundle(tmp_path), ref=REF, version="0.0.0") + assert q.read_bytes() == JOINED and aider.read_bytes() == JOINED + assert {c: disk()["installed_skills"][c] for c in extra} == extra + + def test_moved_home_record_pruned(self, tmp_path): + gone = tmp_path / "old-home" / ".claude" / "commands" / "deepgram" / "api.md" + seed({claude("docs"): b"mine"}, record=False) + state = disk() + state["installed_skills"] = { + "claude": {"paths": [str(gone), str(claude("docs"))]} + } + sg._STATE_FILE.write_text(json.dumps(state), encoding="utf-8") + install(tmp_path) + assert str(gone) not in legacy_paths("claude") + + +class TestShared: + @pytest.mark.parametrize("cli", sorted(SHARED)) + def test_block_removal_keeps_user_text(self, tmp_path, capsys, cli): + seed_text = b"# My notes\n\nKeep this.\n" + data = V032.shared({n: BLOB[n].decode() for n in NAMES}, seed_text.decode()) + path = at(SHARED[cli]) + seed({path: data.encode()}) + install(tmp_path, cli) + assert path.read_bytes() == seed_text # cmp-equal to the pre-0.3.x file. + text = err(capsys) + assert f"INFO: Removed the deepctl 0.3.x section from {path}; the rest" in text + assert "Removed deepctl 0.3.x files" not in text + assert legacy_paths(cli) == folder_paths(cli, sorted(NAMES)) + assert "v03" not in disk()["skill_folders"][cli] + + def test_shared_cleanup_then_second_install_is_silent( + self, tmp_path, capsys, monkeypatch + ): + path = at(SHARED["gemini"]) + seed({path: b"user\n\n" + BLOCK}) + install(tmp_path, "gemini") + assert path.read_bytes() == b"user\n" + capsys.readouterr() + calls = [] + real = sg._update_state + monkeypatch.setattr( + sg, "_update_state", lambda *a, **k: (calls.append(a[1]), real(*a, **k)) + ) + install(tmp_path, "gemini") + assert err(capsys) == "" + assert calls == ["E9", "E9b"] # The cleanup wrote nothing. + assert "v03" not in disk()["skill_folders"]["gemini"] + + def test_shared_file_without_markers_is_untouched_and_silent( + self, tmp_path, capsys, monkeypatch + ): + path = at(SHARED["gemini"]) + seed({path: b"user text only\n"}, record=False) + before = ident(path) + calls = [] + real = sg._update_state + monkeypatch.setattr( + sg, "_update_state", lambda *a, **k: (calls.append(a[1]), real(*a, **k)) + ) + install(tmp_path, "gemini") + assert ident(path) == before + assert err(capsys) == "" + assert calls == ["E9", "E9b"] + + def test_recorded_file_without_markers_is_done_and_silent(self, tmp_path, capsys): + path = at(SHARED["gemini"]) + seed({path: b"I removed the section myself.\n"}) + before = ident(path) + install(tmp_path, "gemini") + assert ident(path) == before + assert err(capsys) == "" + assert legacy_paths("gemini") == folder_paths("gemini", sorted(NAMES)) + assert "v03" not in disk()["skill_folders"]["gemini"] + + @pytest.mark.parametrize( + ("data", "want"), + [ + (BLOCK, None), # Block only: 0.3.x created the file, so it is deleted. + (b"user\n\n" + BLOCK + b"after\n", b"user\nafter\n"), + (crlf(b"user\n\n" + BLOCK + b"after\n"), b"user\r\nafter\r\n"), + (b"\xef\xbb\xbfuser\n\n" + BLOCK, b"\xef\xbb\xbfuser\n"), # BOM kept. + (b"\n\n" + BLOCK, b"\n"), # An empty user file: one newline left. + (b"user\n\n\n" + BLOCK, b"user\n\n\n"), # Not 0.3.x's shape: kept as is. + (b"user" + b"\n" + BLOCK, b"user\n"), # No blank line before it. + ], + ids=["only", "after", "crlf", "bom", "empty", "three-eol", "no-blank"], + ) + def test_block_shapes(self, tmp_path, data, want): + path = at(SHARED["codex"]) + seed({path: data}) + install(tmp_path, "codex") + assert (path.read_bytes() if path.exists() else None) == want + assert legacy_paths("codex") == folder_paths("codex", sorted(NAMES)) + + @pytest.mark.parametrize( + "data", + [ + BLOCK + BLOCK, # Duplicate. + BLOCK[: BLOCK.index(b"\n") + 1] + BLOCK, # Nested BEGIN. + BLOCK[: -len(sg._V03_END) - 1], # Unterminated. + b"a\n" + sg._V03_END + b"\nx\n" + sg._V03_BEGIN + b"\n", # END first. + b"text " + BLOCK, # BEGIN not on its own line. + BLOCK[:-1] + b" trailing\n", # END not on its own line. + b"user\r\n\r\n" + BLOCK, # Mixed line endings. + b"```\n" + BLOCK + b"```\n" + BLOCK, # A second copy in a code fence. + ], + ids=["dup", "nested", "open", "end-first", "begin-inline", "end-inline"] + + ["mixed-eol", "fenced"], + ) + def test_incomplete_repeated_or_mixed_sections_kept(self, tmp_path, capsys, data): + path = at(SHARED["opencode"]) + seed({path: data}) + before = ident(path) + install(tmp_path, "opencode") + assert ident(path) == before + text = err(capsys) + assert "WARN: deepctl can't safely remove its 0.3.x section from" in text + assert "delete it" not in text # Never for a file with the user's text. + assert legacy_paths("opencode") == folder_paths("opencode", sorted(NAMES)) + install(tmp_path, "opencode") + assert err(capsys) == "" + + @POSIX + def test_hard_link_and_foreign_owner_kept(self, tmp_path, capsys, monkeypatch): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + os.link(path, tmp_path / "other-name") + install(tmp_path, "gemini") + assert path.read_bytes() == b"u\n\n" + BLOCK + assert "it has other hard links or another user owns it" in err(capsys) + os.unlink(tmp_path / "other-name") + seed({path: b"u\n\n" + BLOCK}) + monkeypatch.setattr(os, "getuid", lambda: os.stat(path).st_uid + 1) + install(tmp_path, "gemini") + assert path.read_bytes() == b"u\n\n" + BLOCK + assert "it has other hard links or another user owns it" in err(capsys) + + @POSIX + def test_mode_times_and_line_endings_kept(self, tmp_path): + path = at(SHARED["gemini"]) + seed({path: crlf(b"u\n\n" + BLOCK)}) + path.chmod(0o640) + os.utime(path, ns=(1_600_000_000_000_000_000, 1_600_000_000_123_456_789)) + install(tmp_path, "gemini") + assert path.read_bytes() == b"u\r\n" + assert stat.S_IMODE(os.stat(path).st_mode) == 0o640 + assert os.stat(path).st_mtime_ns == 1_600_000_000_123_456_789 + + @pytest.mark.parametrize("lock", ["read-only", "root", "uchg"]) + def test_read_only_or_locked_file_kept_once( + self, tmp_path, capsys, monkeypatch, lock + ): + if lock == "root": # os.access lets root write anything: the mode decides. + monkeypatch.setattr(os, "access", lambda *a, **k: True) + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + try: + if lock == "uchg": + try: + os.chflags(path, stat.UF_IMMUTABLE) + except (AttributeError, OSError) as exc: + pytest.skip(f"no chflags uchg here: {exc}") + else: + path.chmod(0o444) + install(tmp_path, "gemini") + why = "it is read-only or locked" + assert note("E34", path=path, why=why, what=sg._V03_LINES) in err(capsys) + install(tmp_path, "gemini") + assert err(capsys) == "" + finally: + with contextlib.suppress(AttributeError, OSError): + os.chflags(path, 0) + path.chmod(0o644) + assert path.read_bytes() == b"u\n\n" + BLOCK + assert [n for n in os.listdir(path.parent) if n.startswith(sg._V03_ASIDE)] == [] + + def test_windows_branch_leaves_legacy_content_for_manual_removal( + self, tmp_path, capsys, monkeypatch + ): + monkeypatch.setattr(sg, "_WINDOWS", True) + monkeypatch.delattr(os, "getuid", raising=False) + shared, cursor = at(SHARED["gemini"]), at(STANDALONE["cursor"]) + seed({shared: b"u\n\n" + BLOCK, cursor: JOINED}) + monkeypatch.setattr( + sg, + "_v03_file", + lambda *a: pytest.fail("Windows must not mutate legacy paths"), + ) + install(tmp_path, "gemini") + install(tmp_path, "cursor") + assert shared.read_bytes() == b"u\n\n" + BLOCK + assert cursor.read_bytes() == JOINED + assert "does not clean 0.3.x content on Windows" in err(capsys) + + def test_concurrent_change_keeps_file_and_drops_temp( + self, tmp_path, capsys, monkeypatch + ): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + real = sg._read_regular + + def read(p, limit, fd=None): # The re-proof reads the moved file. + data = real(p, limit, fd) + return b"changed" if Path(p).name.startswith(sg._V03_ASIDE) else data + + monkeypatch.setattr(sg, "_read_regular", read) + install(tmp_path, "gemini") + assert path.read_bytes() == b"u\n\n" + BLOCK + assert "it changed while deepctl was editing it" in err(capsys) + assert [n for n in os.listdir(path.parent) if n.startswith(sg._V03_ASIDE)] == [] + + def test_section_line_reports_a_completed_cut(self, tmp_path, capsys): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + install(tmp_path, "gemini") + text = err(capsys) + assert f"Removed the deepctl 0.3.x section from {path}" in text + assert "Removed deepctl 0.3.x files" not in text + + def test_e35_on_replace_failure(self, tmp_path, capsys, monkeypatch): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + real = sg._rename_excl + + def rename(src, dest, fd=None): # The publish: the temp onto the name. + if Path(dest).name == path.name and Path(src).name.endswith(".tmp"): + raise PermissionError(13, "Permission denied") + return real(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", rename) + install(tmp_path, "gemini") + assert path.read_bytes() == b"u\n\n" + BLOCK + assert note("E35", path=path, reason="Permission denied") in err(capsys) + assert [n for n in os.listdir(path.parent) if n.startswith(sg._V03_ASIDE)] == [] + assert legacy_paths("gemini") == [str(path)] + + def test_block_fixture_regenerates_from_v032_writer(self, tmp_path): + block = V032.shared({n: BLOB[n].decode() for n in NAMES}, None).encode() + assert block == BLOCK + assert hashlib.sha256(block).hexdigest().startswith("b6158ef6") + seed({at(SHARED["gemini"]): block}) + install(tmp_path, "gemini") + assert not at(SHARED["gemini"]).exists() + + def test_gate_keeps_section_when_a_folder_did_not_land(self, tmp_path, capsys): + seed({at(SHARED["gemini"]): b"u\n\n" + BLOCK}) + install(tmp_path, "gemini", ("docs", "setup-mcp", "starters")) + assert at(SHARED["gemini"]).read_bytes() == b"u\n\n" + BLOCK + assert err(capsys) == "" + assert legacy_paths("gemini") == [str(at(SHARED["gemini"]))] + + @pytest.mark.parametrize("dangling", [False, True]) + def test_link_kept(self, tmp_path, capsys, dangling): + link_kept(tmp_path, capsys, "gemini", at(SHARED["gemini"]), BLOCK, dangling) + + def test_temp_prefix_is_not_staging(self, tmp_path, monkeypatch): + seed({at(SHARED["codex"]): b"u\n\n" + BLOCK}) + names, real = [], os.open + + def open_(p, *a, **k): + names.append(Path(p).name) + return real(p, *a, **k) + + monkeypatch.setattr(os, "open", open_) + install(tmp_path, "codex") + assert [n for n in names if n.startswith(sg._V03_ASIDE) and n.endswith(".tmp")] + assert at(SHARED["codex"]).read_bytes() == b"u\n" + + +class TestLinksAndKinds: + @pytest.mark.parametrize("dangling", [False, True]) + @pytest.mark.parametrize("kind", ["claude", "cursor"]) + def test_link_at_legacy_path_kept(self, tmp_path, capsys, kind, dangling): + path = {"claude": claude("api"), "cursor": at(STANDALONE["cursor"])}[kind] + data = {"claude": BLOB["api"], "cursor": JOINED}[kind] + link_kept(tmp_path, capsys, kind, path, data, dangling) + + def test_directory_at_legacy_path_kept(self, tmp_path, capsys): + seed({claude("api"): b""}) + claude("api").unlink() + claude("api").mkdir() + install(tmp_path) + assert claude("api").is_dir() + assert "(it is not a file)" in err(capsys) + + def test_oversized_file_kept(self, tmp_path, capsys, monkeypatch): + monkeypatch.setattr(sg, "_V03_MAX", 10) + seed({claude("api"): BLOB["api"]}) + install(tmp_path) + assert claude("api").read_bytes() == BLOB["api"] + assert "(it is larger than 16 MiB)" in err(capsys) + + def test_a_file_that_changes_before_the_read_is_not_called_too_large( + self, tmp_path, capsys, monkeypatch + ): + seed({claude("api"): BLOB["api"]}) + real = sg._read_regular + + def read(p, n, fd=None): # A save between the lstat and the read. + return None if Path(p).name == "api.md" else real(p, n, fd) + + monkeypatch.setattr(sg, "_read_regular", read) + install(tmp_path) + text = err(capsys) + assert "(it changed while deepctl read it)" in text + assert "larger than 16 MiB" not in text + assert claude("api").read_bytes() == BLOB["api"] + + @POSIX + @pytest.mark.skipif( + hasattr(os, "geteuid") and os.geteuid() == 0, reason="root reads anything" + ) + @pytest.mark.parametrize("record", [True, False], ids=["recorded", "unrecorded"]) + def test_unreadable_parent_is_e35_only_if_recorded(self, tmp_path, capsys, record): + seed({claude("api"): BLOB["api"]}, record=record) + claude("api").parent.chmod(0) + try: + install(tmp_path) + text = err(capsys) + finally: + claude("api").parent.chmod(0o755) + assert ("Could not remove deepctl 0.3.x content from" in text) is record + assert ("is unchanged and still recorded" in text) is record # E35, not E35b + assert (str(claude("api")) in legacy_paths("claude")) is record + assert claude("api").read_bytes() == BLOB["api"] + + +class TestFailures: + @pytest.mark.parametrize("how", ["swap", "swap-last", "E1", "E9b", "E27"]) + def test_v03_untouched_when_install_fails(self, tmp_path, monkeypatch, how): + files = {claude(n): BLOB[n] for n in NAMES} + seed(files) + state = disk() + state["skill_folders"] = {"claude": {"folders": {}, "v03": True}} + sg._STATE_FILE.write_text(json.dumps(state), encoding="utf-8") + before = {p: ident(p) for p in files} + if how == "swap": + monkeypatch.setattr( + sg, "_swap", _raise(OSError(28, "No space left on device")) + ) + elif how == "swap-last": # Three folders land, then the install fails. + real_swap = sg._swap + + def swap(g, name, *a): + if name == "starters": + raise OSError(28, "No space left on device") + return real_swap(g, name, *a) + + monkeypatch.setattr(sg, "_swap", swap) + elif how == "E1": + (gen("claude").skills_root() / "api").mkdir(parents=True) + elif how == "E9b": + real = sg._update_state + + def update(mutate, failure="E9c", g=None): + if failure == "E9b": + raise sg._err("E9b", g, reason="disk full") + return real(mutate, failure, g) + + monkeypatch.setattr(sg, "_update_state", update) + else: + monkeypatch.setattr(sg, "_LOCK_TIMEOUT", 0.0) + monkeypatch.setattr(sg, "_try_lock", lambda lock: -1) + with pytest.raises(sg.SkillInstallError): + install(tmp_path) + assert {p: ident(p) for p in files} == before + if how != "E27": + assert disk()["installed_skills"]["claude"]["paths"] == [ + str(p) for p in files + ] + + def test_e35_on_retained_copy_failure_install_succeeds( + self, tmp_path, capsys, monkeypatch + ): + seed({claude("api"): BLOB["api"]}) + real = sg._rename_excl + + def rename(src, dest, fd=None): + if Path(dest).name.startswith(".deepctl-kept-v03-"): + raise PermissionError(13, "Permission denied", str(dest)) + return real(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", rename) + placed, _ = install(tmp_path) + assert len(placed) == 4 + assert claude("api").read_bytes() == BLOB["api"] # Put back. + text = err(capsys) + assert note("E35", path=claude("api"), reason="Permission denied") in text + assert f"{claude('api')} is unchanged and still recorded" in text + assert str(claude("api")) in legacy_paths("claude") + assert disk()["skill_folders"]["claude"]["v03"] is True + + def test_e36_when_the_hook_fails(self, tmp_path, capsys, monkeypatch): + seed({claude("api"): BLOB["api"]}) + real = sg._clean_v03 + + def clean(g, root): + monkeypatch.setattr( + sg, "get_skills_state", _raise(sg._err("E8", reason="I/O error")) + ) + real(g, root) + + monkeypatch.setattr(sg, "_clean_v03", clean) + placed, _ = install(tmp_path) + assert len(placed) == 4 + text = err(capsys) + reason = sg._msg("E8", reason="I/O error").rstrip(".") + assert note("E36", display="Claude Code", reason=reason) in text + assert claude("api").read_bytes() == BLOB["api"] + + def test_reproof_fails_after_move_puts_it_back(self, tmp_path, capsys, monkeypatch): + seed({claude("api"): BLOB["api"]}) + real = sg._read_regular + monkeypatch.setattr( + sg, + "_read_regular", + lambda p, n, fd=None: ( + b"other" if Path(p).name.startswith(sg._V03_ASIDE) else real(p, n, fd) + ), + ) + install(tmp_path) + assert claude("api").read_bytes() == BLOB["api"] + assert "it changed while deepctl was removing it" in err(capsys) + assert os.listdir(claude("api").parent) == ["api.md"] + + @pytest.mark.parametrize("key", ["E4", "E37"]) + def test_e4_or_e37_when_put_back_fails(self, tmp_path, capsys, monkeypatch, key): + seed({claude("api"): BLOB["api"]}) + real_read, real_rename = sg._read_regular, sg._rename_excl + monkeypatch.setattr( + sg, + "_read_regular", + lambda p, n, fd=None: ( + b"other" + if Path(p).name.startswith(sg._V03_ASIDE) + else real_read(p, n, fd) + ), + ) + + def rename(src, dest, fd=None): # E37: something was saved at the name. + if Path(src).name.startswith(sg._V03_ASIDE) and key == "E37": + raise FileExistsError(17, "File exists") + if Path(src).name.startswith(sg._V03_ASIDE): + raise PermissionError(13, "Permission denied") + return real_rename(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", rename) + install(tmp_path) + text = err(capsys) + aside = claude("api").with_name(sg._V03_ASIDE + "api.md") + assert note(key, dest=claude("api"), aside=aside) in text + if key == "E37": # Moving it back by hand would replace the save. + assert "move it back by hand" not in text + assert aside.read_bytes() == BLOB["api"] + + @pytest.mark.parametrize("fails_back", [False, True]) + def test_ctrl_c_during_the_move_puts_it_back( + self, tmp_path, capsys, monkeypatch, fails_back + ): + seed({claude("api"): BLOB["api"]}) + real_read, real_rename = sg._read_regular, sg._rename_excl + + def read(p, n, fd=None): + if Path(p).name.startswith(sg._V03_ASIDE): + raise KeyboardInterrupt + return real_read(p, n, fd) + + def rename(src, dest, fd=None): + if fails_back and Path(src).name.startswith(sg._V03_ASIDE): + raise FileExistsError(17, "File exists") + return real_rename(src, dest, fd) + + monkeypatch.setattr(sg, "_read_regular", read) + monkeypatch.setattr(sg, "_rename_excl", rename) + with pytest.raises(KeyboardInterrupt): + install(tmp_path) + text = err(capsys) + if fails_back: # A file is at the name now: compare, never move back over it. + assert "was saved while deepctl was removing its 0.3.x content" in text + assert "move it back by hand" not in text + else: + assert claude("api").read_bytes() == BLOB["api"] + assert "WARN" not in text + + @POSIX + def test_an_aside_swapped_for_a_link_is_never_put_back( + self, tmp_path, capsys, monkeypatch + ): + seed({claude("api"): BLOB["api"]}) + aside = claude("api").with_name(sg._V03_ASIDE + "api.md") + victim, real = tmp_path / "victim", sg._read_regular + victim.write_bytes(b"secret") + + def read(p, n, fd=None): # Another process swaps the aside during the re-proof. + if Path(p).name.startswith(sg._V03_ASIDE): + aside.unlink() + aside.symlink_to(victim) + return b"changed" + return real(p, n, fd) + + monkeypatch.setattr(sg, "_read_regular", read) + install(tmp_path) + assert not claude("api").exists() and not claude("api").is_symlink() + assert aside.is_symlink() and victim.read_bytes() == b"secret" + text = err(capsys) + assert note("E41", dest=claude("api"), aside=aside) in text + assert "move it back by hand" not in text # E4 would point at the link. + assert "left it in place" not in text and "is still recorded" not in text + + def test_a_successful_cleanup_uses_a_nonrecovery_backup_name( + self, tmp_path, capsys + ): + seed({claude("api"): BLOB["api"]}) + install(tmp_path) + backups = kept(claude("api").parent) + assert not claude("api").exists() and len(backups) == 1 + assert not (claude("api").parent / (sg._V03_ASIDE + "api.md")).exists() + assert "kept the original file" in err(capsys) + + @POSIX + def test_after_e41_a_restored_file_gets_e42_not_e38( + self, tmp_path, capsys, monkeypatch + ): + seed({claude("api"): BLOB["api"]}) + aside = claude("api").with_name(sg._V03_ASIDE + "api.md") + victim, real = tmp_path / "victim", sg._read_regular + victim.write_bytes(b"secret") + + def read(p, n, fd=None): # Another process swaps the aside during the re-proof. + if Path(p).name == aside.name: + aside.unlink() + aside.symlink_to(victim) + return b"changed" + return real(p, n, fd) + + monkeypatch.setattr(sg, "_read_regular", read) + install(tmp_path) + assert note("E41", dest=claude("api"), aside=aside) in err(capsys) + monkeypatch.setattr(sg, "_read_regular", real) + claude("api").write_bytes(b"restored") # The user restores it from a backup. + for _ in range(2): # Each run, until the user deletes the aside. + install(tmp_path) + text = err(capsys) + assert note("E42", dest=claude("api"), aside=aside) in text + assert "holds an earlier version" not in text # E38 is for its own files. + assert claude("api").read_bytes() == b"restored" and aside.is_symlink() + assert victim.read_bytes() == b"secret" + + def test_aside_prefix_is_not_staging(self, tmp_path, monkeypatch): + assert sg._V03_ASIDE == ".deepctl-v03-" + assert not sg._V03_ASIDE.startswith(sg._STAGING_PREFIX) + seed({claude("api"): BLOB["api"]}) + names, real = [], sg._rename_excl + + def rename(src, dest, fd=None): + names.append(Path(dest).name) + return real(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", rename) + install(tmp_path) + assert [n for n in names if n.startswith(sg._V03_ASIDE)] == [ + sg._V03_ASIDE + "api.md" # The same name each run: a leftover is found. + ] + assert not claude("api").exists() + + +class TestOutput: + def test_everything_on_stderr_outside_agentic_mode(self, tmp_path, capsys): + output._output_config.update(agentic=False) + seed({claude("api"): BLOB["api"], claude("docs"): b"mine"}) + install(tmp_path) + out, error = capsys.readouterr() + assert out == "" + assert "Removed deepctl 0.3.x files for Claude Code" in error + assert "deepctl can't prove it wrote" in error + + +def _raise(exc): + def fail(*a, **k): + raise exc + + return fail + + +def asides(folder): + names = os.listdir(folder) if folder.is_dir() else [] # Claude's may be gone. + return sorted(n for n in names if n.startswith(sg._V03_ASIDE)) + + +def kept(folder): + names = os.listdir(folder) if folder.is_dir() else [] + return sorted(n for n in names if n.startswith(".deepctl-kept-v03-")) + + +def open_fds(): + return len( + os.listdir("/proc/self/fd" if os.path.isdir("/proc/self/fd") else "/dev/fd") + ) + + +def relink(tmp_path, rel): + """Move ~/rel elsewhere and put a link to it at ~/rel.""" + link, real = at(rel), tmp_path / "elsewhere" + link.rename(real) + try: + link.symlink_to(real, target_is_directory=True) + except (OSError, NotImplementedError) as exc: + pytest.skip(f"cannot create a symlink here: {exc}") + return link, real + + +CASES = { # cli: (its 0.3.x file, what it holds, the folders a link can replace) + "claude": (".claude/commands/deepgram/api.md", BLOB["api"], 3), + "cursor": (STANDALONE["cursor"], JOINED, 2), + "cline": (STANDALONE["cline"], JOINED, 2), + **{cli: (rel, b"u\n\n" + BLOCK, 1) for cli, rel in SHARED.items()}, +} +STEPS = {cli: ["aside", "proof"] for cli in CASES} +STEPS.update({cli: ["temp", "aside", "proof", "publish"] for cli in SHARED}) +DONE = {cli: b"u\n" if cli in SHARED else None for cli in CASES} # Once it's done. +# What the file and its aside hold right after a step is interrupted: +# "legacy" (0.3.x's bytes), "done" (DONE), or None (nothing there). +CTRL_C = { + "before": ("legacy", None), + "temp": ("legacy", None), + "aside": ("legacy", None), + "proof": ("legacy", None), + "publish": ("done", "legacy"), +} +KILL = { + "temp": ("legacy", None), + "aside": (None, "legacy"), + "proof": (None, "legacy"), + "publish": ("done", "legacy"), +} +AFTER_KILL = {"aside": "E39", "proof": "E39", "publish": "E38"} # The next run. + + +def holds(path): + """What ``path`` and its aside hold, or None.""" + aside = path.with_name(sg._V03_ASIDE + path.name) + return tuple(p.read_bytes() if p.exists() else None for p in (path, aside)) + + +def expect(cli, row): + return tuple({"legacy": CASES[cli][1], "done": DONE[cli]}.get(x) for x in row) + + +def hooks(monkeypatch, cli, step, act): + """Call ``act`` once, right after ``step`` of the move-aside protocol (or before + the move for "before"); every call after that is the real one.""" + path, fired = at(CASES[cli][0]), [] + name = path.name + real_ren, real_read = sg._rename_excl, sg._read_regular + real_unlink, real_fsync = os.unlink, os.fsync + + def once(now): + if now and not fired: + fired.append(step) + act() + + def ren(src, dest, fd=None): + moving = Path(src).name == name and Path(dest).name.startswith(sg._V03_ASIDE) + once(step == "before" and moving) + real_ren(src, dest, fd) + once(step == "aside" and moving) + once(step == "publish" and Path(src).name.endswith(".tmp")) + + def read(p, limit, fd=None): + data = real_read(p, limit, fd) + once(step == "proof" and Path(p).name.startswith(sg._V03_ASIDE)) + return data + + def unlink(p, *a, **k): + real_unlink(p, *a, **k) + once(step == "unlink" and Path(p).name == sg._V03_ASIDE + name) + + def fsync(fd): + real_fsync(fd) + once(step == "temp" and any(n.endswith(".tmp") for n in asides(path.parent))) + + monkeypatch.setattr(sg, "_rename_excl", ren) + monkeypatch.setattr(sg, "_read_regular", read) + monkeypatch.setattr(os, "unlink", unlink) + monkeypatch.setattr(os, "fsync", fsync) + return fired + + +class TestLinkedFolders: + @POSIX + @pytest.mark.parametrize( + "cli,depth", [(c, i) for c, (_, _, n) in CASES.items() for i in range(1, n + 1)] + ) + def test_a_link_at_any_folder_keeps_the_file(self, tmp_path, capsys, cli, depth): + rel, data, _ = CASES[cli] + path = at(rel) + seed({path: data}) + link, real = relink(tmp_path, "/".join(rel.split("/")[:depth])) + legacy = real.joinpath(*rel.split("/")[depth:]) + before = open_fds() + install(tmp_path, cli) + assert open_fds() == before + assert link.is_symlink() and legacy.read_bytes() == data + assert not list(real.rglob(sg._V03_ASIDE + "*")) + why = f"{link} is a link, which deepctl doesn't follow" + text, what = err(capsys), "that content yourself" + if cli in SHARED: # Only the marker lines, as E34 says. + what = "only the lines from '' yourself and keep the rest of the file" + assert note("E40", path=path, why=sg._V03Link(0, why), what=what) in text + assert str(path) not in legacy_paths(cli) + install(tmp_path, cli) + assert err(capsys) == "" + assert legacy.read_bytes() == data + + @POSIX + @pytest.mark.parametrize("cli", ["cursor", "gemini"]) + def test_home_itself_a_link_still_cleans(self, tmp_path, monkeypatch, cli): + real = Path.home() + alias = tmp_path / "home-link" + alias.symlink_to(real, target_is_directory=True) + monkeypatch.setattr(Path, "home", staticmethod(lambda: alias)) + seed({at(CASES[cli][0]): CASES[cli][1]}) + install(tmp_path, cli) + monkeypatch.setattr(Path, "home", staticmethod(lambda: real)) + assert holds(at(CASES[cli][0])) == expect(cli, ("done", None)) + + @POSIX + @pytest.mark.parametrize("cli", ["cursor", "gemini"]) + def test_a_folder_swapped_for_a_link_after_the_walk_is_not_followed( + self, tmp_path, monkeypatch, cli + ): + rel, data, _ = CASES[cli] + top = rel.split("/")[0] + seed({at(rel): data}) + victim, moved = tmp_path / "victim", tmp_path / "moved" + (victim / rel).parent.mkdir(parents=True) + (victim / rel).write_bytes(data) + real_walk = sg._V03Dir.walk + + def walk(d): + real_walk(d) + if d.parts[0] == top and not moved.exists(): # Right after the check. + at(top).rename(moved) + at(top).symlink_to(victim / top, target_is_directory=True) + + monkeypatch.setattr(sg._V03Dir, "walk", walk) + install(tmp_path, cli) + assert (victim / rel).read_bytes() == data + assert os.listdir((victim / rel).parent) == [Path(rel).name] + at(top).unlink() + moved.rename(at(top)) + assert holds(at(rel)) == expect( + cli, ("done", None) + ) # In the folder it checked. + + @POSIX + @pytest.mark.parametrize("cli", ["cursor", "gemini"]) + def test_a_folder_swapped_for_a_link_between_its_check_and_open_fails_closed( + self, tmp_path, capsys, monkeypatch, cli + ): + rel, data, _ = CASES[cli] + top = rel.split("/")[0] + seed({at(rel): data}) + victim, moved = tmp_path / "victim", tmp_path / "moved" + (victim / rel).parent.mkdir(parents=True) + (victim / rel).write_bytes(data) + real = os.lstat + + def lstat(p, *a, **k): + st = real(p, *a, **k) + if Path(p) == at(top) and not moved.exists(): # Checked: now swap it. + at(top).rename(moved) + at(top).symlink_to(victim / top, target_is_directory=True) + return st + + monkeypatch.setattr(os, "lstat", lstat) + install(tmp_path, cli) + assert (victim / rel).read_bytes() == data # O_NOFOLLOW: never opened. + assert os.listdir((victim / rel).parent) == [Path(rel).name] + assert moved.joinpath(*rel.split("/")[1:]).read_bytes() == data + + def test_a_file_in_place_of_a_folder_is_left_alone(self, tmp_path, capsys): + path = at(STANDALONE["cursor"]) + seed({path: JOINED}) + shutil.rmtree(path.parent) + path.parent.write_bytes(b"mine") + install(tmp_path, "cursor") + assert path.parent.read_bytes() == b"mine" + assert err(capsys) == "" + + def test_a_failed_publish_puts_it_back_even_if_the_temp_is_gone( + self, tmp_path, capsys, monkeypatch + ): + path, real = at(SHARED["gemini"]), sg._rename_excl + seed({path: b"u\n\n" + BLOCK}) + + def ren(src, dest, fd=None): # The temp vanishes, then the publish fails. + if str(src).endswith(".tmp"): + os.remove(path.parent / Path(src).name) + raise OSError(5, "Input/output error") + return real(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", ren) + install(tmp_path, "gemini") + assert holds(path) == (b"u\n\n" + BLOCK, None) + text = err(capsys) + assert f"{path} is unchanged and still recorded" in text + assert legacy_paths("gemini") == [str(path)] + + @pytest.mark.skipif(os.name == "nt", reason="the Windows branch uses os.rename") + def test_rename_excl_names_relative_to_the_cwd_or_a_folder_fd( + self, tmp_path, monkeypatch + ): + (tmp_path / "cwd").mkdir() + monkeypatch.chdir(tmp_path / "cwd") + Path("a").write_bytes(b"a") + Path("c").write_bytes(b"c") + sg._rename_excl("a", "b") # AT_FDCWD: -2 on macOS, -100 on Linux. + with pytest.raises(FileExistsError): + sg._rename_excl("b", "c") + fd = os.open(tmp_path / "cwd", os.O_RDONLY) + try: + sg._rename_excl("b", "d", fd) + with pytest.raises(FileExistsError): + sg._rename_excl("d", "c", fd) + finally: + os.close(fd) + assert sorted(os.listdir()) == ["c", "d"] + assert Path("d").read_bytes() == b"a" and Path("c").read_bytes() == b"c" + + +class TestInterrupted: + @pytest.mark.parametrize( + "cli,step", [(c, s) for c in CASES for s in ["before", *STEPS[c]]] + ) + def test_ctrl_c_at_each_step_loses_nothing( + self, tmp_path, capsys, monkeypatch, cli, step + ): + path = at(CASES[cli][0]) + seed({path: CASES[cli][1]}) + + def ctrl_c(): + raise KeyboardInterrupt + + fired = hooks(monkeypatch, cli, step, ctrl_c) + with pytest.raises(KeyboardInterrupt): + install(tmp_path, cli) + assert fired == [step] + text = err(capsys) + assert holds(path) == expect(cli, CTRL_C[step]) + assert not [n for n in asides(path.parent) if n.endswith(".tmp")] + aside = path.with_name(sg._V03_ASIDE + path.name) + assert note("E4", dest=path, aside=aside) not in text # Published: E38 next. + assert note("E37", dest=path, aside=aside) not in text # Nor a put-back try. + install(tmp_path, cli) # The next run finishes, or names what it kept. + assert holds(path) == expect(cli, ("done", CTRL_C[step][1])) + assert (note("E38", dest=path, aside=aside) in err(capsys)) is ( + step == "publish" + ) + + @POSIX + @pytest.mark.parametrize("sig", ["SIGKILL", "SIGTERM", "SIGHUP"]) + @pytest.mark.parametrize("cli,step", [(c, s) for c in CASES for s in STEPS[c]]) + def test_killed_at_each_step_the_next_run_recovers( + self, tmp_path, capsys, monkeypatch, sig, cli, step + ): + path, signum = at(CASES[cli][0]), getattr(signal, sig) + aside = path.with_name(sg._V03_ASIDE + path.name) + seed({path: CASES[cli][1]}) + pid = os.fork() + if pid == 0: # The child dies at the step: no except or finally runs. + try: + if signum != signal.SIGKILL: # Which can't have a handler. + signal.signal(signum, signal.SIG_DFL) + hooks(monkeypatch, cli, step, lambda: os.kill(os.getpid(), signum)) + install(tmp_path, cli) + finally: + os._exit(3) + _, status = os.waitpid(pid, 0) + assert os.WIFSIGNALED(status) and os.WTERMSIG(status) == signum + capsys.readouterr() + assert holds(path) == expect(cli, KILL[step]) + stale = [n for n in asides(path.parent) if n.endswith(".tmp")] + assert len(stale) == ( + step in ("temp", "aside", "proof") and "temp" in STEPS[cli] + ) + install(tmp_path, cli) + text = err(capsys) + assert holds(path) == expect( + cli, ("done", KILL[step][1] if step == "publish" else None) + ) + temps = [n for n in asides(path.parent) if n.endswith(".tmp")] + assert temps == stale # The README says what a leftover temp is. + for key in ("E38", "E39"): + assert (note(key, dest=path, aside=aside) in text) is ( + AFTER_KILL.get(step) == key + ) + + def test_a_file_and_its_aside_are_both_kept_and_named_each_run( + self, tmp_path, capsys + ): + path = at(STANDALONE["cursor"]) + aside = path.with_name(sg._V03_ASIDE + path.name) + seed({path: JOINED}) + aside.write_bytes(b"earlier") + for _ in range(2): + install(tmp_path, "cursor") + assert note("E38", dest=path, aside=aside) in err(capsys) + assert path.read_bytes() == JOINED and aside.read_bytes() == b"earlier" + + def test_the_put_back_never_replaces_a_file_that_appears( + self, tmp_path, capsys, monkeypatch + ): + path = at(STANDALONE["cursor"]) + aside = path.with_name(sg._V03_ASIDE + path.name) + seed({path: JOINED}) + path.rename(aside) + real, refused = sg._rename_excl, [] + + def ren(src, dest, fd=None): + if Path(src).name == aside.name: + path.write_bytes(b"new") # Saved right before the put-back. + try: + real(src, dest, fd) + except OSError as exc: # The OS's own text: Windows words it differently. + refused.append(exc) + raise + + monkeypatch.setattr(sg, "_rename_excl", ren) + install(tmp_path, "cursor") + assert path.read_bytes() == b"new" and aside.read_bytes() == JOINED + text = err(capsys) + assert note("E35b", path=path, reason=sg._reason(refused[0])) in text + assert "is unchanged" not in text + assert legacy_paths("cursor") == [str(path)] + + def test_a_failed_put_back_of_an_interrupted_run_is_e35b( + self, tmp_path, capsys, monkeypatch + ): + path = at(STANDALONE["cursor"]) + aside = path.with_name(sg._V03_ASIDE + path.name) + seed({path: JOINED}) + path.rename(aside) # A killed run left it here. + real = sg._rename_excl + + def ren(src, dest, fd=None): + if Path(src).name == aside.name: + raise PermissionError(13, "Permission denied") + real(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", ren) + install(tmp_path, "cursor") + assert not path.exists() and aside.read_bytes() == JOINED + text = err(capsys) + assert note("E35b", path=path, reason="Permission denied") in text + assert "is unchanged" not in text and "put it back" not in text + assert legacy_paths("cursor") == [str(path)] + + @pytest.mark.parametrize("kind", ["folder", pytest.param("link", marks=POSIX)]) + def test_an_aside_that_is_not_a_file_is_not_put_back(self, tmp_path, capsys, kind): + path = at(STANDALONE["cursor"]) + aside = path.with_name(sg._V03_ASIDE + path.name) + seed({path: JOINED}) + target = tmp_path / "target" + path.rename(target) + if kind == "folder": + aside.mkdir() + else: + aside.symlink_to(target) + install(tmp_path, "cursor") + assert not path.exists() and target.read_bytes() == JOINED + assert aside.is_dir() if kind == "folder" else aside.is_symlink() + assert err(capsys) == "" + + +class TestSharedRaces: + @POSIX + @pytest.mark.parametrize("step", ["temp", "proof"]) + def test_a_temp_swapped_for_a_link_is_never_followed_or_published( + self, tmp_path, capsys, monkeypatch, step + ): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + victim = tmp_path / "victim" + victim.write_bytes(b"secret") + victim.chmod(0o600) + os.utime(victim, ns=(1_500_000_000_000_000_000, 1_500_000_000_000_000_000)) + before = os.stat(victim) + + def swap(): # Right before the chmod ("temp") or the publish ("proof"). + (tmp,) = [n for n in asides(path.parent) if n.endswith(".tmp")] + os.remove(path.parent / tmp) + (path.parent / tmp).symlink_to(victim) + + hooks(monkeypatch, "gemini", step, swap) + install(tmp_path, "gemini") + after = os.stat(victim) + assert (after.st_mode, after.st_mtime_ns) == ( + before.st_mode, + before.st_mtime_ns, + ) + assert victim.read_bytes() == b"secret" + assert not path.is_symlink() and holds(path) == (b"u\n\n" + BLOCK, None) + assert asides(path.parent) == [] # The swapped-in link is gone too. + text = err(capsys) # GEMINI.md itself never changed. + assert "(deepctl's temporary copy of it was replaced)" in text + assert "it changed while deepctl was editing it" not in text + + def test_a_save_between_the_proof_and_the_publish_is_kept( + self, tmp_path, capsys, monkeypatch + ): + path = at(SHARED["gemini"]) + aside = path.with_name(sg._V03_ASIDE + path.name) + seed({path: b"u\n\n" + BLOCK}) + + def save(): # An editor saves atomically: a temp file renamed over the name. + (path.parent / "editor.swp").write_bytes(b"EDITOR\n") + os.replace(path.parent / "editor.swp", path) + + hooks(monkeypatch, "gemini", "proof", save) + install(tmp_path, "gemini") + text = err(capsys) + assert holds(path) == (b"EDITOR\n", b"u\n\n" + BLOCK) + assert asides(path.parent) == [aside.name] # No temp left. + assert note("E37", dest=path, aside=aside) in text + assert "move it back by hand" not in text # E4 would undo the save. + assert "can't safely remove its 0.3.x section" not in text # E34 contradicts. + install(tmp_path, "gemini") + assert note("E38", dest=path, aside=aside) in err(capsys) + assert holds(path) == (b"EDITOR\n", b"u\n\n" + BLOCK) + + @POSIX + @pytest.mark.parametrize( + ("cli", "path", "data", "active"), + [ + ("cursor", STANDALONE["cursor"], JOINED, None), + ("gemini", SHARED["gemini"], b"u\n\n" + BLOCK, b"u\n"), + ], + ids=["standalone", "shared"], + ) + def test_a_write_through_an_open_descriptor_stays_recoverable( + self, tmp_path, capsys, cli, path, data, active + ): + path = at(path) + seed({path: data}) + with open(path, "r+b") as writer: + install(tmp_path, cli) + writer.seek(0) + writer.write(b"LATE") + writer.flush() + os.fsync(writer.fileno()) + backups = kept(path.parent) + assert (path.read_bytes() if path.exists() else None) == active + assert len(backups) == 1 + assert (path.parent / backups[0]).read_bytes().startswith(b"LATE") + assert "kept the original file" in err(capsys) + + @POSIX + def test_a_refused_publish_puts_it_back_and_leaves_no_fd( + self, tmp_path, capsys, monkeypatch + ): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + before, real = open_fds(), sg._rename_excl + + def ren(src, dest, fd=None): + if Path(src).name.endswith(".tmp"): + raise PermissionError(13, "Permission denied") + return real(src, dest, fd) + + monkeypatch.setattr(sg, "_rename_excl", ren) + install(tmp_path, "gemini") + assert open_fds() == before + assert holds(path) == (b"u\n\n" + BLOCK, None) and asides(path.parent) == [] + assert note("E35", path=path, reason="Permission denied") in err(capsys) + assert legacy_paths("gemini") == [str(path)] + monkeypatch.setattr(sg, "_rename_excl", real) + install(tmp_path, "gemini") + assert open_fds() == before + assert holds(path) == (b"u\n", None) + + def test_a_full_disk_while_writing_the_temp_leaves_the_file( + self, tmp_path, capsys, monkeypatch + ): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + + def full(): + raise OSError(28, "No space left on device") + + hooks(monkeypatch, "gemini", "temp", full) + install(tmp_path, "gemini") + assert holds(path) == (b"u\n\n" + BLOCK, None) and asides(path.parent) == [] + reason = "No space left on device" + text = err(capsys) + assert note("E35", path=path, reason=reason) in text + assert f"{path} is unchanged and still recorded" in text + assert legacy_paths("gemini") == [str(path)] + + def test_a_published_cut_keeps_the_original_copy(self, tmp_path, capsys): + path = at(SHARED["gemini"]) + seed({path: b"u\n\n" + BLOCK}) + install(tmp_path, "gemini") + text = err(capsys) + backups = kept(path.parent) + assert path.read_bytes() == b"u\n" + assert len(backups) == 1 + assert (path.parent / backups[0]).read_bytes() == b"u\n\n" + BLOCK + assert f"Removed the deepctl 0.3.x section from {path}" in text + assert "kept the original file" in text + + def test_a_removed_section_only_file_keeps_the_original_copy( + self, tmp_path, capsys + ): + path = at(SHARED["codex"]) + seed({path: BLOCK}) + install(tmp_path, "codex") + backups = kept(path.parent) + assert not path.exists() + assert len(backups) == 1 + assert (path.parent / backups[0]).read_bytes() == BLOCK + assert "kept the original file" in err(capsys)