Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,8 @@ Providers: Deepgram, AssemblyAI, OpenAI, Cartesia, ElevenLabs, MiniMax, Speechif

OpenAI STT supports `whisper-1`, `gpt-4o-transcribe`, and `gpt-4o-mini-transcribe` through the independent `openai.stt_model` setting.

Telnyx STT fronts a dozen engines behind one key (`telnyx.transcription_engine`, default `Deepgram` so barge-in and live captions work out of the box); the in-house `Telnyx` engine is finals-only, so both are off when it is selected, and a startup log line says so.

## Documentation

| Page | What's in it |
Expand Down
2 changes: 2 additions & 0 deletions README.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,8 @@ StreamCore 位于「提示词 + 工具」类框架的下一层:媒体链路。

OpenAI STT 可通过独立的 `openai.stt_model` 配置选择 `whisper-1`、`gpt-4o-transcribe` 或 `gpt-4o-mini-transcribe`。

Telnyx STT 用一个 key 前置十余种引擎(`telnyx.transcription_engine`,默认 `Deepgram`,打断与实时字幕开箱即用);自研 `Telnyx` 引擎只出最终结果,选中它时两者关闭,启动日志会写明。

## 文档

| 页面 | 内容 |
Expand Down
9 changes: 7 additions & 2 deletions config.toml.example
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ turn_merge_ms = 350 # Debounce window for merging finals into one turn.
provider = "" # Supported: grok (empty = classic pipeline)

[stt]
provider = "deepgram" # Supported: aliyun, assemblyai, deepgram, openai, vibevoice, volcengine
provider = "deepgram" # Supported: aliyun, assemblyai, deepgram, openai, telnyx, vibevoice, volcengine

[llm]
provider = "openai" # Supported: openai, ollama, agent
Expand Down Expand Up @@ -143,11 +143,16 @@ voice_id = "" # Optional; defaults to Geffen (geffen_32). Simba 3.2 us
model = "" # Optional; defaults to simba-3.2

[telnyx]
api_key = "" # Required if tts.provider = "telnyx"
api_key = "" # Required if tts.provider = "telnyx" or stt.provider = "telnyx"
voice = "Telnyx.Qwen3TTS.d9348e0d-988a-42cc-a64e-18093fe45c03" # Any catalog voice from GET /v2/text-to-speech/voices
# (Qwen3TTS voices use UUID ids). Availability varies by account; a voice your
# key is not provisioned for fails the dial with HTTP 403
voice_speed = 1.0 # Playback-rate multiplier, clamped to 0.8-1.2 like the per-utterance delivery tags
transcription_engine = "Deepgram" # STT. Transcription engine when stt.provider = "telnyx".
# Verified values: "Deepgram" (streams partials, so barge-in and live captions
# work) or "Telnyx" (in-house, finals-only, so barge-in and live captions are
# off and a startup log line says so). Case-sensitive, sent verbatim; other
# hosted engines the endpoint fronts pass through untested

[mimo]
api_key = "" # Required if tts.provider = "mimo"
Expand Down
7 changes: 5 additions & 2 deletions docs/configuration.md
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ turn_merge_ms = 350 # Debounce window for merging finals into o
provider = "" # "grok", or empty for the classic pipeline

[stt]
provider = "deepgram" # aliyun | assemblyai | deepgram | openai | vibevoice | volcengine
provider = "deepgram" # aliyun | assemblyai | deepgram | openai | telnyx | vibevoice | volcengine

[llm]
provider = "openai" # openai | ollama | agent
Expand Down Expand Up @@ -119,10 +119,13 @@ api_key = ""
voice_id = ""
model = ""

[telnyx] # Telnyx hosted synthesis, used when tts.provider = "telnyx"
[telnyx] # Telnyx hosted speech, used when tts.provider = "telnyx" or stt.provider = "telnyx"; one key covers both roles
api_key = ""
voice = "Telnyx.Qwen3TTS.d9348e0d-988a-42cc-a64e-18093fe45c03" # Any catalog voice from GET /v2/text-to-speech/voices; availability varies by account
voice_speed = 1.0 # Playback-rate multiplier, clamped to 0.8-1.2
transcription_engine = "Deepgram" # STT engine, verified values: "Deepgram" (partials, barge-in works; the default) or
# "Telnyx" (in-house, finals-only: barge-in and live captions are off, and a startup
# log line says so). Case-sensitive, sent verbatim

[minimax]
api_key = ""
Expand Down
6 changes: 4 additions & 2 deletions docs/configuration.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@ turn_merge_ms = 350 # Debounce window for merging finals into o
provider = "" # "grok", or empty for the classic pipeline

[stt]
provider = "deepgram" # aliyun | assemblyai | deepgram | openai | vibevoice | volcengine
provider = "deepgram" # aliyun | assemblyai | deepgram | openai | telnyx | vibevoice | volcengine

[llm]
provider = "openai" # openai | ollama | agent
Expand Down Expand Up @@ -113,10 +113,12 @@ api_key = ""
voice_id = ""
model = ""

[telnyx] # Telnyx 托管合成,当 tts.provider = "telnyx" 时使用
[telnyx] # Telnyx 托管语音,当 tts.provider = "telnyx" 或 stt.provider = "telnyx" 时使用;一个 key 覆盖两个方向
api_key = ""
voice = "Telnyx.Qwen3TTS.d9348e0d-988a-42cc-a64e-18093fe45c03" # GET /v2/text-to-speech/voices 目录中的任意音色;可用性因账号而异
voice_speed = 1.0 # 播放速率倍数,限制在 0.8-1.2
transcription_engine = "Deepgram" # STT 引擎,已验证取值:"Deepgram"(有中间结果,打断可用;默认值)或
# "Telnyx"(自研,只出最终结果:打断与实时字幕关闭,启动日志会写明)。大小写敏感,原样透传

[minimax]
api_key = ""
Expand Down
26 changes: 24 additions & 2 deletions docs/providers.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,15 +4,16 @@

| Role | Providers | Required credentials |
|------|-----------|----------------------|
| STT | `aliyun`, `assemblyai`, `deepgram`, `openai`, `vibevoice`, `volcengine` | Matching provider API key, or a local VibeVoice ASR server |
| STT | `aliyun`, `assemblyai`, `deepgram`, `openai`, `telnyx`, `vibevoice`, `volcengine` | Matching provider API key, or a local VibeVoice ASR server |
| LLM | `openai`, `ollama`, `agent` | OpenAI API key, an Ollama instance you control, or your own HTTP agent endpoint |
| TTS | `cartesia`, `deepgram`, `elevenlabs`, `mimo`, `minimax`, `speechify`, `telnyx`, `vibevoice` | Matching provider API key, or a local VibeVoice TTS server |
| Speech-to-speech | `grok` | xAI API key — replaces STT, LLM, and TTS together |
| RAG (optional) | `pgvector`, `supabase` | Postgres connection string or Supabase URL + key, plus an OpenAI key for embeddings |

Notes:

- `stt.provider = "openai"` uses batch final transcription instead of streaming partials; choose `whisper-1`, `gpt-4o-transcribe`, or `gpt-4o-mini-transcribe` with `openai.stt_model`.
- `stt.provider = "openai"` uses batch final transcription instead of streaming partials, so barge-in and live captions do not work, since both depend on partials; choose `whisper-1`, `gpt-4o-transcribe`, or `gpt-4o-mini-transcribe` with `openai.stt_model`.
- `stt.provider = "telnyx"` fronts Telnyx's in-house and a dozen hosted transcription engines over one WebSocket and one key. `transcription_engine` defaults to `Deepgram`, so barge-in and live captions work out of the box; the in-house `Telnyx` engine is finals-only, so both are off when it is selected. See [Telnyx STT](#telnyx-stt).
- `llm.provider = "ollama"` targets any Ollama-compatible endpoint via `base_url` — local or on your own infrastructure.
- `llm.provider = "agent"` POSTs each turn to an HTTP endpoint you host; your agent owns memory, prompting, and tools, and replies stream back as SSE, chunked text, or JSON. See [Bring your own agent](./bring-your-own-agent.md).
- `stt.provider = "vibevoice"` and `tts.provider = "vibevoice"` use local models; start the Python sidecars first.
Expand Down Expand Up @@ -148,6 +149,27 @@ Three things to know:

Delivery tags map onto `voice_speed` (clamped to 0.8–1.2, the same conversational band as Cartesia), and `voice_speed` in config sets the baseline pace for untagged sentences.

## Telnyx STT

Streaming transcription over the Telnyx speech-to-text WebSocket: raw linear16 binary frames in, JSON transcript frames out, at the pipeline's native 16 kHz mono so nothing resamples. The same `[telnyx]` section and API key as TTS cover both roles; `transcription_engine` picks the recognizer.

```toml
[stt]
provider = "telnyx"

[telnyx]
api_key = ""
transcription_engine = "Deepgram" # verified: "Deepgram" (partial results, barge-in works) or "Telnyx" (in-house, finals-only). Case matters
```

Three things to know:

- **`Deepgram` is the default.** It streams interim results exactly like the built-in Deepgram provider, so barge-in and live captions work out of the box. A finals-only default would silently switch both off for anyone who just sets the provider and starts talking.
- **The in-house `Telnyx` engine is finals-only.** It emits exactly one final per utterance, only after the caller stops speaking: no interims, no timestamps, confidence `null`. Barge-in and live captions depend on interim results (the pipeline gates interruption on partial text), so **neither works with this engine**: interruptions never fire, and the client transcript shows nothing until the final lands. Because the engine answers only after audio stops, the client runs its own endpointing (`internal/vad`, the same detector the pipeline uses, with the silence timeout `openai.go` uses) and opens one socket per utterance, closing it once the final lands, since the server keeps it open. A startup log line says that barge-in and live captions are off for the session, so the operator learns it from the log rather than from a caller talking over the agent with nothing happening. See the [design discussion](https://github.com/streamcoreai/streamcore-server/issues/75) for the trade-offs.
- **The engine name is case-sensitive and sent verbatim.** `telnyx` is rejected with a structured error frame that lists the supported engines. Only `Deepgram` and `Telnyx` are verified here; the other hosted engines the endpoint fronts (AssemblyAI, Azure, and the rest of that list) pass through untested.

Confidence arrives as `null` from the in-house engine and a 0-1 float from hosted ones; the pipeline treats `null` as unknown rather than low.

## Local VibeVoice setup

VibeVoice provides fully local STT and TTS with no API keys, using [VibeVoice-ASR](https://huggingface.co/mlx-community/VibeVoice-ASR-4bit) for recognition and [VibeVoice-Realtime-0.5B](https://huggingface.co/mlx-community/VibeVoice-Realtime-0.5B-6bit) for synthesis via two lightweight Python sidecars. On Apple Silicon they use [mlx-audio](https://github.com/Blaizzy/mlx-audio) (MLX); on Linux/Windows they fall back to PyTorch automatically.
Expand Down
26 changes: 24 additions & 2 deletions docs/providers.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,15 +4,16 @@

| 角色 | 服务商 | 所需凭据 |
|------|-----------|----------------------|
| STT | `aliyun`、`assemblyai`、`deepgram`、`openai`、`vibevoice`、`volcengine` | 对应服务商的 API key,或一个本地 VibeVoice ASR 服务 |
| STT | `aliyun`、`assemblyai`、`deepgram`、`openai`、`telnyx`、`vibevoice`、`volcengine` | 对应服务商的 API key,或一个本地 VibeVoice ASR 服务 |
| LLM | `openai`、`ollama`、`agent` | OpenAI API key、你自己掌控的 Ollama 实例,或你自己的 HTTP 智能体端点 |
| TTS | `cartesia`、`deepgram`、`elevenlabs`、`mimo`、`minimax`、`speechify`、`telnyx`、`vibevoice` | 对应服务商的 API key,或一个本地 VibeVoice TTS 服务 |
| 语音到语音 | `grok` | xAI API key —— 一并取代 STT、LLM 与 TTS |
| RAG(可选) | `pgvector`、`supabase` | Postgres 连接串或 Supabase URL + key,另需 OpenAI key 用于 embedding |

注意:

- `stt.provider = "openai"` 使用批量最终转写而不是流式中间结果;可通过 `openai.stt_model` 选择 `whisper-1`、`gpt-4o-transcribe` 或 `gpt-4o-mini-transcribe`。
- `stt.provider = "openai"` 使用批量最终转写而不是流式中间结果,打断(barge-in)与实时字幕都依赖中间结果,因此都不工作;可通过 `openai.stt_model` 选择 `whisper-1`、`gpt-4o-transcribe` 或 `gpt-4o-mini-transcribe`。
- `stt.provider = "telnyx"` 通过一条 WebSocket、一个 key 前置自研与十余种托管转写引擎。`transcription_engine` 默认 `Deepgram`,打断与实时字幕开箱即用;自研 `Telnyx` 引擎只出最终结果,选中它时两者关闭。见 [Telnyx STT](#telnyx-stt)。
- `llm.provider = "ollama"` 通过 `base_url` 指向任何兼容 Ollama 的端点 —— 本地或你自己的基础设施均可。
- `llm.provider = "agent"` 把每一轮对话 POST 到你托管的 HTTP 端点;记忆、提示词与工具都由你的智能体掌控,回复以 SSE、分块文本或 JSON 流式返回。见[接入你自己的智能体](./bring-your-own-agent.zh-CN.md)。
- `stt.provider = "vibevoice"` 与 `tts.provider = "vibevoice"` 使用本地模型;请先启动 Python 边车进程。
Expand Down Expand Up @@ -148,6 +149,27 @@ voice_speed = 1.0

表达标签映射到 `voice_speed`(限制在 0.8–1.2,与 Cartesia 相同的对话档位),配置里的 `voice_speed` 则是未打标签句子的基准语速。

## Telnyx STT

走 Telnyx 语音转文字 WebSocket 的流式识别:上行是裸 linear16 二进制帧,下行是 JSON 转写帧,均为流水线原生的 16 kHz 单声道,音频路径无需任何重采样。与 TTS 共用同一个 `[telnyx]` 配置段和 API key;`transcription_engine` 选择识别引擎。

```toml
[stt]
provider = "telnyx"

[telnyx]
api_key = ""
transcription_engine = "Deepgram" # 已验证取值:"Deepgram"(有中间结果,打断可用)或 "Telnyx"(自研,只出最终结果)。大小写敏感
```

有三件事必须弄对:

- **默认是 `Deepgram`。** 它像内置的 Deepgram 服务商一样流式输出中间结果,打断(barge-in)与实时字幕开箱即用。一个只出最终结果的默认引擎,会让任何只改了 provider 就开始说话的人悄无声息地失去这两项能力。
- **自研 `Telnyx` 引擎只出最终结果。** 它在来电者停止说话后才发出唯一一帧 final —— 没有中间结果、没有时间戳、置信度为 `null`。打断与实时字幕依赖中间结果(流水线以部分文本来判定打断),因此**这两项在该引擎下不工作**:打断永远不会触发,客户端字幕也要等到 final 落地才有内容。因为该引擎只在整个话语结束后才应答,客户端自己做端点检测(`internal/vad`,即流水线在用的同一个检测器,静音窗口与 `openai.go` 相同),每个话语开一条新连接,final 落地后由客户端关闭(服务端会一直握着连接不放)。启动时日志里会写明本会话的打断与实时字幕已关闭,运维从日志就能知道,而不用等到来电者对着智能体说话却毫无反应。取舍讨论见[设计讨论](https://github.com/streamcoreai/streamcore-server/issues/75)。
- **引擎名大小写敏感,原样透传。** `telnyx` 会被一帧结构化错误拒绝,错误里列出支持的引擎。此处只验证了 `Deepgram` 与 `Telnyx` 两个取值;该端点前置的其他托管引擎(AssemblyAI、Azure 及列表中的其余引擎)可透传但未经测试。

置信度:自研引擎返回 `null`,托管引擎返回 0-1 浮点数;流水线把 `null` 视为未知而不是低置信。

## 本地 VibeVoice 配置

VibeVoice 提供完全本地、无需 API key 的 STT 与 TTS:识别用 [VibeVoice-ASR](https://huggingface.co/mlx-community/VibeVoice-ASR-4bit),合成用 [VibeVoice-Realtime-0.5B](https://huggingface.co/mlx-community/VibeVoice-Realtime-0.5B-6bit),通过两个轻量 Python 边车进程运行。在 Apple Silicon 上使用 [mlx-audio](https://github.com/Blaizzy/mlx-audio)(MLX);在 Linux/Windows 上自动回退到 PyTorch。
Expand Down
14 changes: 12 additions & 2 deletions internal/config/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -293,8 +293,8 @@ type SpeechifyConfig struct {
Model string `toml:"model"`
}

// TelnyxConfig configures Telnyx hosted speech synthesis, used when
// tts.provider = "telnyx".
// TelnyxConfig configures Telnyx hosted speech, used when tts.provider =
// "telnyx" and when stt.provider = "telnyx". One API key covers both roles.
type TelnyxConfig struct {
APIKey string `toml:"api_key"`
// Voice is a catalog voice from GET /v2/text-to-speech/voices. Empty
Expand All @@ -305,6 +305,15 @@ type TelnyxConfig struct {
// VoiceSpeed is a playback-rate multiplier (1.0 = normal), clamped to
// the 0.8-1.2 conversational band. Zero defaults to 1.0.
VoiceSpeed float64 `toml:"voice_speed"`
// TranscriptionEngine selects the recognizer used when stt.provider =
// "telnyx". The value is case-sensitive and sent to the endpoint
// verbatim. "Deepgram" (the default) streams partials, so barge-in and
// live captions work out of the box; the in-house "Telnyx" engine
// emits one final per utterance and no interims, so barge-in and live
// captions do not work with it and a startup log line says so. Other
// engines the endpoint fronts (AssemblyAI, Azure, ...) pass through
// untested.
TranscriptionEngine string `toml:"transcription_engine"`
}

type MiMoConfig struct {
Expand Down Expand Up @@ -452,6 +461,7 @@ func Load(path string) (*Config, error) {
if cfg.Telnyx.VoiceSpeed == 0 {
cfg.Telnyx.VoiceSpeed = 1.0
}
setDefault(&cfg.Telnyx.TranscriptionEngine, "Deepgram")
setDefault(&cfg.LLM.Provider, "openai")
setDefault(&cfg.TTS.Provider, "cartesia")
setDefault(&cfg.OpenAI.Model, "gpt-4o-mini")
Expand Down
7 changes: 6 additions & 1 deletion internal/stt/stt.go
Original file line number Diff line number Diff line change
Expand Up @@ -51,9 +51,14 @@ func NewClient(ctx context.Context, cfg *config.Config, onResult func(Transcript
return nil, fmt.Errorf("stt provider %q requires [volcengine] api_key to be set", cfg.STT.Provider)
}
return NewVolcengineClient(ctx, cfg.Volcengine, onResult)
case "telnyx":
if cfg.Telnyx.APIKey == "" {
return nil, fmt.Errorf("stt provider %q requires [telnyx] api_key to be set", cfg.STT.Provider)
}
return NewTelnyxClient(ctx, cfg.Telnyx, onResult)
case "vibevoice":
return NewVibeVoiceClient(ctx, cfg.VibeVoice.ASRURL, onResult)
default:
return nil, fmt.Errorf("unknown stt provider %q (supported: aliyun, assemblyai, deepgram, openai, vibevoice, volcengine)", cfg.STT.Provider)
return nil, fmt.Errorf("unknown stt provider %q (supported: aliyun, assemblyai, deepgram, openai, telnyx, vibevoice, volcengine)", cfg.STT.Provider)
}
}
Loading
Loading