AsyncAPI YAML
asyncapi: 2.6.0
info:
title: REST API
version: 1.0.0
channels:
/v1/agent/converse:
description: Build a conversational voice agent using Deepgram's Voice Agent WebSocket
bindings:
ws:
headers:
type: object
properties:
Authorization:
type: string
publish:
operationId: subpackage_agent/v1.agent.v1-publish
summary: Server messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-0-AgentV1ListenUpdated'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-1-AgentV1ThinkUpdated'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-2-AgentV1ReceiveFunctionCallResponse'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-3-AgentV1PromptUpdated'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-4-AgentV1SpeakUpdated'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-5-AgentV1InjectionRefused'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-6-AgentV1Welcome'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-7-AgentV1SettingsApplied'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-8-AgentV1ConversationText'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-9-AgentV1UserStartedSpeaking'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-10-AgentV1AgentThinking'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-11-AgentV1LatencyReport'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-12-AgentV1FunctionCallRequest'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-13-AgentV1FunctionCallCancelled'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-14-AgentV1AgentStartedSpeaking'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-15-AgentV1AgentAudioDone'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-16-AgentV1Error'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-17-AgentV1Warning'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-18-AgentV1History'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-server-19-AgentV1Audio'
subscribe:
operationId: subpackage_agent/v1.agent.v1-subscribe
summary: Client messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-0-AgentV1Settings'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-1-AgentV1UpdateListen'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-2-AgentV1UpdateThink'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-3-AgentV1UpdateSpeak'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-4-AgentV1InjectUserMessage'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-5-AgentV1InjectAgentMessage'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-6-AgentV1SendFunctionCallResponse'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-7-AgentV1KeepAlive'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-8-AgentV1UpdatePrompt'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-9-AgentV1ForceEndTurn'
- $ref: '#/components/messages/subpackage_agent/v1.agent.v1-client-10-AgentV1Media'
/v1/listen:
description: Transcribe audio and video using Deepgram's speech-to-text WebSocket
bindings:
ws:
query:
type: object
properties:
callback:
$ref: '#/components/schemas/ListenV1Callback'
callback_method:
$ref: '#/components/schemas/ListenV1CallbackMethod'
default: POST
channels:
$ref: '#/components/schemas/ListenV1Channels'
default: '1'
detect_entities:
$ref: '#/components/schemas/ListenV1DetectEntities'
default: 'false'
diarize:
$ref: '#/components/schemas/ListenV1Diarize'
default: 'false'
diarize_model:
$ref: '#/components/schemas/ListenV1_diarize_model'
dictation:
$ref: '#/components/schemas/ListenV1Dictation'
default: 'false'
encoding:
$ref: '#/components/schemas/ListenV1Encoding'
endpointing:
$ref: '#/components/schemas/ListenV1Endpointing'
default: '10'
extra:
$ref: '#/components/schemas/ListenV1Extra'
interim_results:
$ref: '#/components/schemas/ListenV1InterimResults'
default: 'false'
keyterm:
$ref: '#/components/schemas/ListenV1Keyterm'
keywords:
$ref: '#/components/schemas/ListenV1Keywords'
language:
$ref: '#/components/schemas/ListenV1Language'
default: en
mip_opt_out:
$ref: '#/components/schemas/ListenV1MipOptOut'
default: 'false'
model:
$ref: '#/components/schemas/ListenV1Model'
multichannel:
$ref: '#/components/schemas/ListenV1Multichannel'
default: 'false'
numerals:
$ref: '#/components/schemas/ListenV1Numerals'
default: 'false'
profanity_filter:
$ref: '#/components/schemas/ListenV1ProfanityFilter'
default: 'false'
punctuate:
$ref: '#/components/schemas/ListenV1Punctuate'
default: 'false'
redact:
$ref: '#/components/schemas/ListenV1Redact'
default: 'false'
replace:
$ref: '#/components/schemas/ListenV1Replace'
sample_rate:
$ref: '#/components/schemas/ListenV1SampleRate'
search:
$ref: '#/components/schemas/ListenV1Search'
smart_format:
$ref: '#/components/schemas/ListenV1SmartFormat'
default: 'false'
tag:
$ref: '#/components/schemas/ListenV1Tag'
utterance_end_ms:
$ref: '#/components/schemas/ListenV1UtteranceEndMs'
vad_events:
$ref: '#/components/schemas/ListenV1VadEvents'
default: 'false'
version:
$ref: '#/components/schemas/ListenV1Version'
default: latest
headers:
type: object
properties:
Authorization:
type: string
publish:
operationId: subpackage_listen/v1.listen.v1-publish
summary: Server messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-server-0-ListenV1Results'
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-server-1-ListenV1Metadata'
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-server-2-ListenV1UtteranceEnd'
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-server-3-ListenV1SpeechStarted'
subscribe:
operationId: subpackage_listen/v1.listen.v1-subscribe
summary: Client messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-client-0-ListenV1Media'
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-client-1-ListenV1Finalize'
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-client-2-ListenV1CloseStream'
- $ref: '#/components/messages/subpackage_listen/v1.listen.v1-client-3-ListenV1KeepAlive'
/v2/listen:
description: |
Real-time conversational speech recognition with contextual turn detection
for natural voice conversations
bindings:
ws:
query:
type: object
properties:
model:
$ref: '#/components/schemas/ListenV2Model'
encoding:
$ref: '#/components/schemas/ListenV2Encoding'
sample_rate:
$ref: '#/components/schemas/ListenV2SampleRate'
eager_eot_threshold:
$ref: '#/components/schemas/ListenV2EagerEotThreshold'
eot_threshold:
$ref: '#/components/schemas/ListenV2EotThreshold'
default: '0.7'
eot_timeout_ms:
$ref: '#/components/schemas/ListenV2EotTimeoutMs'
default: '5000'
keyterm:
$ref: '#/components/schemas/ListenV2Keyterm'
language_hint:
$ref: '#/components/schemas/ListenV2LanguageHint'
profanity_filter:
$ref: '#/components/schemas/ListenV2ProfanityFilter'
default: 'false'
numerals:
$ref: '#/components/schemas/ListenV2Numerals'
default: 'false'
redact:
$ref: '#/components/schemas/ListenV2Redact'
mip_opt_out:
$ref: '#/components/schemas/ListenV2MipOptOut'
tag:
$ref: '#/components/schemas/ListenV2Tag'
headers:
type: object
properties:
Authorization:
type: string
publish:
operationId: subpackage_listen/v2.listen.v2-publish
summary: Server messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-server-0-ListenV2Connected'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-server-1-ListenV2TurnInfo'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-server-2-ListenV2ConfigureSuccess'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-server-3-ListenV2ConfigureFailure'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-server-4-ListenV2Warning'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-server-5-ListenV2FatalError'
subscribe:
operationId: subpackage_listen/v2.listen.v2-subscribe
summary: Client messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-client-0-ListenV2Media'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-client-1-ListenV2CloseStream'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-client-2-ListenV2ForceEndTurn'
- $ref: '#/components/messages/subpackage_listen/v2.listen.v2-client-3-ListenV2Configure'
/v1/speak:
description: Convert text into natural-sounding speech using Deepgram's TTS WebSocket
bindings:
ws:
query:
type: object
properties:
encoding:
$ref: '#/components/schemas/SpeakV1Encoding'
default: linear16
mip_opt_out:
$ref: '#/components/schemas/SpeakV1MipOptOut'
default: 'false'
model:
$ref: '#/components/schemas/SpeakV1Model'
default: aura-asteria-en
sample_rate:
$ref: '#/components/schemas/SpeakV1SampleRate'
default: '24000'
speed:
$ref: '#/components/schemas/SpeakV1Speed'
default: 1
headers:
type: object
properties:
Authorization:
type: string
publish:
operationId: subpackage_speak/v1.speak.v1-publish
summary: Server messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-server-0-SpeakV1Audio'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-server-1-SpeakV1Metadata'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-server-2-SpeakV1Flushed'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-server-3-SpeakV1Cleared'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-server-4-SpeakV1Warning'
subscribe:
operationId: subpackage_speak/v1.speak.v1-subscribe
summary: Client messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-client-0-SpeakV1Text'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-client-1-SpeakV1Flush'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-client-2-SpeakV1Clear'
- $ref: '#/components/messages/subpackage_speak/v1.speak.v1-client-3-SpeakV1Close'
/v2/speak:
description: |
Streaming, turn-based text-to-speech (Flux TTS) built for voice-agent
pipelines. Stream LLM tokens in, speak them to the user, and report
per-turn billing and timing.
bindings:
ws:
query:
type: object
properties:
model:
$ref: '#/components/schemas/SpeakV2Model'
encoding:
$ref: '#/components/schemas/SpeakV2Encoding'
default: linear16
sample_rate:
$ref: '#/components/schemas/SpeakV2SampleRate'
speed:
$ref: '#/components/schemas/SpeakV2Speed'
default: 1
expressivity:
$ref: '#/components/schemas/SpeakV2Expressivity'
default: 0
mip_opt_out:
$ref: '#/components/schemas/SpeakV2MipOptOut'
default: 'false'
tag:
$ref: '#/components/schemas/SpeakV2Tag'
headers:
type: object
properties:
Authorization:
type: string
publish:
operationId: subpackage_speak/v2.speak.v2-publish
summary: Server messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-0-SpeakV2Audio'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-1-SpeakV2Connected'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-2-SpeakV2SpeechStarted'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-3-SpeakV2SpeechMetadata'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-4-SpeakV2SpeechInterrupted'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-5-SpeakV2Flushed'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-6-SpeakV2SessionMetadata'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-7-SpeakV2ConfigureSuccess'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-8-SpeakV2ConfigureFailure'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-9-SpeakV2Warning'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-server-10-SpeakV2Error'
subscribe:
operationId: subpackage_speak/v2.speak.v2-subscribe
summary: Client messages
message:
oneOf:
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-client-0-SpeakV2Speak'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-client-1-SpeakV2Flush'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-client-2-SpeakV2Interrupt'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-client-3-SpeakV2Configure'
- $ref: '#/components/messages/subpackage_speak/v2.speak.v2-client-4-SpeakV2Close'
servers:
Production:
url: wss://api.deepgram.com/
protocol: wss
x-default: true
components:
messages:
subpackage_agent/v1.agent.v1-server-0-AgentV1ListenUpdated:
name: AgentV1ListenUpdated
title: AgentV1ListenUpdated
description: Receive listen update from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1ListenUpdated'
subpackage_agent/v1.agent.v1-server-1-AgentV1ThinkUpdated:
name: AgentV1ThinkUpdated
title: AgentV1ThinkUpdated
description: Receive think update from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1ThinkUpdated'
subpackage_agent/v1.agent.v1-server-2-AgentV1ReceiveFunctionCallResponse:
name: AgentV1ReceiveFunctionCallResponse
title: AgentV1ReceiveFunctionCallResponse
description: |
Receive a function call response from the server after the server
has executed a server-side function call internally. This occurs
when functions are marked with client_side: false.
payload:
$ref: '#/components/schemas/AgentV1_AgentV1ReceiveFunctionCallResponse'
subpackage_agent/v1.agent.v1-server-3-AgentV1PromptUpdated:
name: AgentV1PromptUpdated
title: AgentV1PromptUpdated
description: Receive prompt update from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1PromptUpdated'
subpackage_agent/v1.agent.v1-server-4-AgentV1SpeakUpdated:
name: AgentV1SpeakUpdated
title: AgentV1SpeakUpdated
description: Receive speak update from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1SpeakUpdated'
subpackage_agent/v1.agent.v1-server-5-AgentV1InjectionRefused:
name: AgentV1InjectionRefused
title: AgentV1InjectionRefused
description: Receive injection refused message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1InjectionRefused'
subpackage_agent/v1.agent.v1-server-6-AgentV1Welcome:
name: AgentV1Welcome
title: AgentV1Welcome
description: Receive welcome message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1Welcome'
subpackage_agent/v1.agent.v1-server-7-AgentV1SettingsApplied:
name: AgentV1SettingsApplied
title: AgentV1SettingsApplied
description: Receive settings applied message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1SettingsApplied'
subpackage_agent/v1.agent.v1-server-8-AgentV1ConversationText:
name: AgentV1ConversationText
title: AgentV1ConversationText
description: Receive conversation text from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1ConversationText'
subpackage_agent/v1.agent.v1-server-9-AgentV1UserStartedSpeaking:
name: AgentV1UserStartedSpeaking
title: AgentV1UserStartedSpeaking
description: Receive user started speaking message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1UserStartedSpeaking'
subpackage_agent/v1.agent.v1-server-10-AgentV1AgentThinking:
name: AgentV1AgentThinking
title: AgentV1AgentThinking
description: Receive agent thinking message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1AgentThinking'
subpackage_agent/v1.agent.v1-server-11-AgentV1LatencyReport:
name: AgentV1LatencyReport
title: AgentV1LatencyReport
description: Receive a latency report from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1LatencyReport'
subpackage_agent/v1.agent.v1-server-12-AgentV1FunctionCallRequest:
name: AgentV1FunctionCallRequest
title: AgentV1FunctionCallRequest
description: Receive function call request from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1FunctionCallRequest'
subpackage_agent/v1.agent.v1-server-13-AgentV1FunctionCallCancelled:
name: AgentV1FunctionCallCancelled
title: AgentV1FunctionCallCancelled
description: Receive notice that a function call you already received was cancelled because the user started speaking again
payload:
$ref: '#/components/schemas/AgentV1_AgentV1FunctionCallCancelled'
subpackage_agent/v1.agent.v1-server-14-AgentV1AgentStartedSpeaking:
name: AgentV1AgentStartedSpeaking
title: AgentV1AgentStartedSpeaking
description: Receive agent started speaking message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1AgentStartedSpeaking'
subpackage_agent/v1.agent.v1-server-15-AgentV1AgentAudioDone:
name: AgentV1AgentAudioDone
title: AgentV1AgentAudioDone
description: Receive agent audio done message from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1AgentAudioDone'
subpackage_agent/v1.agent.v1-server-16-AgentV1Error:
name: AgentV1Error
title: AgentV1Error
description: Receive error response from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1Error'
subpackage_agent/v1.agent.v1-server-17-AgentV1Warning:
name: AgentV1Warning
title: AgentV1Warning
description: Receive warning messages from Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1Warning'
subpackage_agent/v1.agent.v1-server-18-AgentV1History:
name: AgentV1History
title: AgentV1History
description: >-
Receive a conversation history message from Deepgram's Voice Agent API. Each message is either a conversation
text (with role and content) or a function call record (with function_calls array).
payload:
$ref: '#/components/schemas/AgentV1_AgentV1History'
subpackage_agent/v1.agent.v1-server-19-AgentV1Audio:
name: AgentV1Audio
title: AgentV1Audio
description: Receive raw binary audio data generated by Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1Audio'
subpackage_agent/v1.agent.v1-client-0-AgentV1Settings:
name: AgentV1Settings
title: AgentV1Settings
description: Send settings configuration to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1Settings'
subpackage_agent/v1.agent.v1-client-1-AgentV1UpdateListen:
name: AgentV1UpdateListen
title: AgentV1UpdateListen
description: Send update listen to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1UpdateListen'
subpackage_agent/v1.agent.v1-client-2-AgentV1UpdateThink:
name: AgentV1UpdateThink
title: AgentV1UpdateThink
description: Send update think to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1UpdateThink'
subpackage_agent/v1.agent.v1-client-3-AgentV1UpdateSpeak:
name: AgentV1UpdateSpeak
title: AgentV1UpdateSpeak
description: Send update speak to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1UpdateSpeak'
subpackage_agent/v1.agent.v1-client-4-AgentV1InjectUserMessage:
name: AgentV1InjectUserMessage
title: AgentV1InjectUserMessage
description: Send inject user message to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1InjectUserMessage'
subpackage_agent/v1.agent.v1-client-5-AgentV1InjectAgentMessage:
name: AgentV1InjectAgentMessage
title: AgentV1InjectAgentMessage
description: Send inject agent message to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1InjectAgentMessage'
subpackage_agent/v1.agent.v1-client-6-AgentV1SendFunctionCallResponse:
name: AgentV1SendFunctionCallResponse
title: AgentV1SendFunctionCallResponse
description: |
Send a function call response from the client to the server after
executing a client-side function call. This is used when the server
requests execution of a function marked with client_side: true.
payload:
$ref: '#/components/schemas/AgentV1_AgentV1SendFunctionCallResponse'
subpackage_agent/v1.agent.v1-client-7-AgentV1KeepAlive:
name: AgentV1KeepAlive
title: AgentV1KeepAlive
description: Send keep alive to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1KeepAlive'
subpackage_agent/v1.agent.v1-client-8-AgentV1UpdatePrompt:
name: AgentV1UpdatePrompt
title: AgentV1UpdatePrompt
description: Send a prompt update to Deepgram's Voice Agent API
payload:
$ref: '#/components/schemas/AgentV1_AgentV1UpdatePrompt'
subpackage_agent/v1.agent.v1-client-9-AgentV1ForceEndTurn:
name: AgentV1ForceEndTurn
title: AgentV1ForceEndTurn
description: Send a ForceEndTurn message to immediately end the current user turn
payload:
$ref: '#/components/schemas/AgentV1_AgentV1ForceEndTurn'
subpackage_agent/v1.agent.v1-client-10-AgentV1Media:
name: AgentV1Media
title: AgentV1Media
description: Send raw binary audio data to Deepgram's Voice Agent API for processing
payload:
$ref: '#/components/schemas/AgentV1_AgentV1Media'
subpackage_listen/v1.listen.v1-server-0-ListenV1Results:
name: ListenV1Results
title: ListenV1Results
description: Receive transcription results
payload:
$ref: '#/components/schemas/ListenV1_ListenV1Results'
subpackage_listen/v1.listen.v1-server-1-ListenV1Metadata:
name: ListenV1Metadata
title: ListenV1Metadata
description: Receive metadata about the transcription
payload:
$ref: '#/components/schemas/ListenV1_ListenV1Metadata'
subpackage_listen/v1.listen.v1-server-2-ListenV1UtteranceEnd:
name: ListenV1UtteranceEnd
title: ListenV1UtteranceEnd
description: Receive an utterance end event
payload:
$ref: '#/components/schemas/ListenV1_ListenV1UtteranceEnd'
subpackage_listen/v1.listen.v1-server-3-ListenV1SpeechStarted:
name: ListenV1SpeechStarted
title: ListenV1SpeechStarted
description: Receive a speech started event
payload:
$ref: '#/components/schemas/ListenV1_ListenV1SpeechStarted'
subpackage_listen/v1.listen.v1-client-0-ListenV1Media:
name: ListenV1Media
title: ListenV1Media
description: Send audio or video data to be transcribed
payload:
$ref: '#/components/schemas/ListenV1_ListenV1Media'
subpackage_listen/v1.listen.v1-client-1-ListenV1Finalize:
name: ListenV1Finalize
title: ListenV1Finalize
description: Send a Finalize message to flush the WebSocket stream
payload:
$ref: '#/components/schemas/ListenV1_ListenV1Finalize'
subpackage_listen/v1.listen.v1-client-2-ListenV1CloseStream:
name: ListenV1CloseStream
title: ListenV1CloseStream
description: Send a CloseStream message to close the WebSocket stream
payload:
$ref: '#/components/schemas/ListenV1_ListenV1CloseStream'
subpackage_listen/v1.listen.v1-client-3-ListenV1KeepAlive:
name: ListenV1KeepAlive
title: ListenV1KeepAlive
description: Send a KeepAlive message to keep the WebSocket stream alive
payload:
$ref: '#/components/schemas/ListenV1_ListenV1KeepAlive'
subpackage_listen/v2.listen.v2-server-0-ListenV2Connected:
name: ListenV2Connected
title: ListenV2Connected
description: Receive a connected message
payload:
$ref: '#/components/schemas/ListenV2_ListenV2Connected'
subpackage_listen/v2.listen.v2-server-1-ListenV2TurnInfo:
name: ListenV2TurnInfo
title: ListenV2TurnInfo
description: Receive a turn info message
payload:
$ref: '#/components/schemas/ListenV2_ListenV2TurnInfo'
subpackage_listen/v2.listen.v2-server-2-ListenV2ConfigureSuccess:
name: ListenV2ConfigureSuccess
title: ListenV2ConfigureSuccess
description: >-
Sent when a Configure message was successfully applied. Returns the current, up-to-date values that were
applied.
payload:
$ref: '#/components/schemas/ListenV2_ListenV2ConfigureSuccess'
subpackage_listen/v2.listen.v2-server-3-ListenV2ConfigureFailure:
name: ListenV2ConfigureFailure
title: ListenV2ConfigureFailure
description: Indicates that a Configure message was rejected
payload:
$ref: '#/components/schemas/ListenV2_ListenV2ConfigureFailure'
subpackage_listen/v2.listen.v2-server-4-ListenV2Warning:
name: ListenV2Warning
title: ListenV2Warning
description: Receive a warning; the server keeps the connection open
payload:
$ref: '#/components/schemas/ListenV2_ListenV2Warning'
subpackage_listen/v2.listen.v2-server-5-ListenV2FatalError:
name: ListenV2FatalError
title: ListenV2FatalError
description: Receive a fatal error message
payload:
$ref: '#/components/schemas/ListenV2_ListenV2FatalError'
subpackage_listen/v2.listen.v2-client-0-ListenV2Media:
name: ListenV2Media
title: ListenV2Media
description: Send audio or video data to be transcribed
payload:
$ref: '#/components/schemas/ListenV2_ListenV2Media'
subpackage_listen/v2.listen.v2-client-1-ListenV2CloseStream:
name: ListenV2CloseStream
title: ListenV2CloseStream
description: Send a CloseStream message to close the WebSocket stream
payload:
$ref: '#/components/schemas/ListenV2_ListenV2CloseStream'
subpackage_listen/v2.listen.v2-client-2-ListenV2ForceEndTurn:
name: ListenV2ForceEndTurn
title: ListenV2ForceEndTurn
description: Send a ForceEndTurn message to immediately end the current turn
payload:
$ref: '#/components/schemas/ListenV2_ListenV2ForceEndTurn'
subpackage_listen/v2.listen.v2-client-3-ListenV2Configure:
name: ListenV2Configure
title: ListenV2Configure
description: Send a Configure message to update Flux settings
payload:
$ref: '#/components/schemas/ListenV2_ListenV2Configure'
subpackage_speak/v1.speak.v1-server-0-SpeakV1Audio:
name: SpeakV1Audio
title: SpeakV1Audio
description: Receive audio chunks as they are generated
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Audio'
subpackage_speak/v1.speak.v1-server-1-SpeakV1Metadata:
name: SpeakV1Metadata
title: SpeakV1Metadata
description: Receive metadata about the audio generation
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Metadata'
subpackage_speak/v1.speak.v1-server-2-SpeakV1Flushed:
name: SpeakV1Flushed
title: SpeakV1Flushed
description: Receive metadata about the audio generation
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Flushed'
subpackage_speak/v1.speak.v1-server-3-SpeakV1Cleared:
name: SpeakV1Cleared
title: SpeakV1Cleared
description: Receive metadata about the audio generation
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Cleared'
subpackage_speak/v1.speak.v1-server-4-SpeakV1Warning:
name: SpeakV1Warning
title: SpeakV1Warning
description: Receive a warning about the audio generation
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Warning'
subpackage_speak/v1.speak.v1-client-0-SpeakV1Text:
name: SpeakV1Text
title: SpeakV1Text
description: Text to convert to audio
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Text'
subpackage_speak/v1.speak.v1-client-1-SpeakV1Flush:
name: SpeakV1Flush
title: SpeakV1Flush
description: Flush the buffer and receive the final audio for text sent so far
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Flush'
subpackage_speak/v1.speak.v1-client-2-SpeakV1Clear:
name: SpeakV1Clear
title: SpeakV1Clear
description: Clear the buffer and start a new audio generation. Potentially destructive operation for any text in the buffer
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Clear'
subpackage_speak/v1.speak.v1-client-3-SpeakV1Close:
name: SpeakV1Close
title: SpeakV1Close
description: Flush the buffer and close the connection gracefully after all audio is generated
payload:
$ref: '#/components/schemas/SpeakV1_SpeakV1Close'
subpackage_speak/v2.speak.v2-server-0-SpeakV2Audio:
name: SpeakV2Audio
title: SpeakV2Audio
description: Receive audio chunks as they are generated
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Audio'
subpackage_speak/v2.speak.v2-server-1-SpeakV2Connected:
name: SpeakV2Connected
title: SpeakV2Connected
description: Receive a connected message on a successful connection
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Connected'
subpackage_speak/v2.speak.v2-server-2-SpeakV2SpeechStarted:
name: SpeakV2SpeechStarted
title: SpeakV2SpeechStarted
description: Receive a message marking the start of a new turn, carrying the turn's unique identifier
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2SpeechStarted'
subpackage_speak/v2.speak.v2-server-3-SpeakV2SpeechMetadata:
name: SpeakV2SpeechMetadata
title: SpeakV2SpeechMetadata
description: Receive per-turn billing and timing after a manual Flush
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2SpeechMetadata'
subpackage_speak/v2.speak.v2-server-4-SpeakV2SpeechInterrupted:
name: SpeakV2SpeechInterrupted
title: SpeakV2SpeechInterrupted
description: Receive what the user heard, and the interrupted turn's billing, after an Interrupt
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2SpeechInterrupted'
subpackage_speak/v2.speak.v2-server-5-SpeakV2Flushed:
name: SpeakV2Flushed
title: SpeakV2Flushed
description: Receive an echo confirming receipt of a manual Flush
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Flushed'
subpackage_speak/v2.speak.v2-server-6-SpeakV2SessionMetadata:
name: SpeakV2SessionMetadata
title: SpeakV2SessionMetadata
description: Receive cumulative session totals before the socket closes
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2SessionMetadata'
subpackage_speak/v2.speak.v2-server-7-SpeakV2ConfigureSuccess:
name: SpeakV2ConfigureSuccess
title: SpeakV2ConfigureSuccess
description: Receive confirmation that a Configure was accepted and applied, echoing the applied configuration
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2ConfigureSuccess'
subpackage_speak/v2.speak.v2-server-8-SpeakV2ConfigureFailure:
name: SpeakV2ConfigureFailure
title: SpeakV2ConfigureFailure
description: Receive notice that a Configure was rejected or failed to apply; the prior configuration is retained
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2ConfigureFailure'
subpackage_speak/v2.speak.v2-server-9-SpeakV2Warning:
name: SpeakV2Warning
title: SpeakV2Warning
description: Receive a warning; synthesis continues and the connection is unaffected
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Warning'
subpackage_speak/v2.speak.v2-server-10-SpeakV2Error:
name: SpeakV2Error
title: SpeakV2Error
description: Receive a fatal error message followed by a WebSocket close
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Error'
subpackage_speak/v2.speak.v2-client-0-SpeakV2Speak:
name: SpeakV2Speak
title: SpeakV2Speak
description: Send text to be synthesized into the active turn
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Speak'
subpackage_speak/v2.speak.v2-client-1-SpeakV2Flush:
name: SpeakV2Flush
title: SpeakV2Flush
description: End the active turn and generate the remaining audio
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Flush'
subpackage_speak/v2.speak.v2-client-2-SpeakV2Interrupt:
name: SpeakV2Interrupt
title: SpeakV2Interrupt
description: Cancel the active turn because the user barged in
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Interrupt'
subpackage_speak/v2.speak.v2-client-3-SpeakV2Configure:
name: SpeakV2Configure
title: SpeakV2Configure
description: Update synthesis configuration mid-session
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Configure'
subpackage_speak/v2.speak.v2-client-4-SpeakV2Close:
name: SpeakV2Close
title: SpeakV2Close
description: Gracefully close the connection, draining all remaining and queued audio
payload:
$ref: '#/components/schemas/SpeakV2_SpeakV2Close'
schemas:
AgentV1_AgentV1ListenUpdated:
type: object
properties:
type:
type: string
enum:
- ListenUpdated
description: Message type identifier for listen update confirmation
required:
- type
title: AgentV1_AgentV1ListenUpdated
AgentV1_AgentV1ThinkUpdated:
type: object
properties:
type:
type: string
enum:
- ThinkUpdated
description: Message type identifier for think update confirmation
required:
- type
title: AgentV1_AgentV1ThinkUpdated
AgentV1_AgentV1ReceiveFunctionCallResponse:
type: object
properties:
type:
type: string
enum:
- FunctionCallResponse
description: Message type identifier for function call responses
id:
type: string
description: |
The unique identifier for the function call.
• **Required for client responses**: Should match the id from
the corresponding `FunctionCallRequest`
• **Optional for server responses**: Server may omit when responding
to internal function executions
name:
type: string
description: The name of the function being called
content:
type: string
description: The content or result of the function call
required:
- type
- name
- content
description: |
Function call response message used bidirectionally:
• **Client → Server**: Response after client executes a function
marked as client_side: true
• **Server → Client**: Response after server executes a function
marked as client_side: false
The same message structure serves both directions, enabling a unified
interface for function call responses regardless of execution location.
title: AgentV1_AgentV1ReceiveFunctionCallResponse
AgentV1_AgentV1PromptUpdated:
type: object
properties:
type:
type: string
enum:
- PromptUpdated
description: Message type identifier for prompt update confirmation
required:
- type
title: AgentV1_AgentV1PromptUpdated
AgentV1_AgentV1SpeakUpdated:
type: object
properties:
type:
type: string
enum:
- SpeakUpdated
description: Message type identifier for speak update confirmation
required:
- type
title: AgentV1_AgentV1SpeakUpdated
AgentV1_AgentV1InjectionRefused:
type: object
properties:
type:
type: string
enum:
- InjectionRefused
description: Message type identifier for injection refused
message:
type: string
description: Details about why the injection was refused
required:
- type
- message
title: AgentV1_AgentV1InjectionRefused
AgentV1_AgentV1Welcome:
type: object
properties:
type:
type: string
enum:
- Welcome
description: Message type identifier for welcome message
request_id:
type: string
description: Unique identifier for the request
required:
- type
- request_id
title: AgentV1_AgentV1Welcome
AgentV1_AgentV1SettingsApplied:
type: object
properties:
type:
type: string
enum:
- SettingsApplied
description: Message type identifier for settings applied confirmation
required:
- type
title: AgentV1_AgentV1SettingsApplied
ChannelsAgentV1MessagesAgentV1ConversationTextRole:
type: string
enum:
- user
- assistant
description: Identifies who spoke the statement
title: ChannelsAgentV1MessagesAgentV1ConversationTextRole
AgentV1_AgentV1ConversationText:
type: object
properties:
type:
type: string
enum:
- ConversationText
description: Message type identifier for conversation text
role:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1ConversationTextRole'
description: Identifies who spoke the statement
content:
type: string
description: The actual statement that was spoken
languages_hinted:
type: array
items:
type: string
description: >-
The language hints that were active at the time of the turn. Only present on user-role messages when the
listen model is flux-general-multi.
languages:
type: array
items:
type: string
description: >-
Languages detected in the user's speech, sorted by word count (descending). Only present on user-role
messages when the listen model is flux-general-multi.
required:
- type
- role
- content
title: AgentV1_AgentV1ConversationText
AgentV1_AgentV1UserStartedSpeaking:
type: object
properties:
type:
type: string
enum:
- UserStartedSpeaking
description: Message type identifier indicating that the user has begun speaking
required:
- type
title: AgentV1_AgentV1UserStartedSpeaking
AgentV1_AgentV1AgentThinking:
type: object
properties:
type:
type: string
enum:
- AgentThinking
description: Message type identifier for agent thinking
content:
type: string
description: The text of the agent's thought process
required:
- type
- content
title: AgentV1_AgentV1AgentThinking
AgentV1_AgentV1LatencyReport:
type: object
properties:
type:
type: string
enum:
- LatencyReport
description: Message type identifier for the latency report
stt_latency:
type: string
title: float
description: 'Speech-to-text: time from audio received to transcript produced, in seconds'
ttt_token_latency:
type: string
title: float
description: Time to first token of any type (text, tool call, or thinking), in seconds
ttt_text_latency:
type: string
title: float
description: Time to first text token from the LLM, in seconds
ttt_tool_latency:
type: string
title: float
description: Time to first tool-call token from the LLM, in seconds
ttt_thinking_latency:
type: string
title: float
description: Time to first thinking token from the LLM, in seconds
tts_latency:
type: string
title: float
description: 'Text-to-speech: time from first text token to first audio byte, in seconds'
total_latency:
type: string
title: float
description: 'End-to-end: time from user utterance end to first audio byte, in seconds'
required:
- type
title: AgentV1_AgentV1LatencyReport
ChannelsAgentV1MessagesAgentV1FunctionCallRequestFunctionsItems:
type: object
properties:
id:
type: string
description: Unique identifier for the function call
name:
type: string
description: The name of the function to call
arguments:
type: string
description: JSON string containing the function arguments
client_side:
type: boolean
description: Whether the function should be executed client-side
thought_signature:
type: string
description: Some Gemini models require this as an additional function call identifier
required:
- id
- name
- arguments
- client_side
title: ChannelsAgentV1MessagesAgentV1FunctionCallRequestFunctionsItems
AgentV1_AgentV1FunctionCallRequest:
type: object
properties:
type:
type: string
enum:
- FunctionCallRequest
description: Message type identifier for function call requests
functions:
type: array
items:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1FunctionCallRequestFunctionsItems'
description: Array of functions to be called
required:
- type
- functions
title: AgentV1_AgentV1FunctionCallRequest
ChannelsAgentV1MessagesAgentV1FunctionCallCancelledFunctionsItems:
type: object
properties:
id:
type: string
description: The id from the FunctionCallRequest that is now cancelled. Send no FunctionCallResponse for this id
name:
type: string
description: The name of the cancelled function
required:
- id
- name
title: ChannelsAgentV1MessagesAgentV1FunctionCallCancelledFunctionsItems
AgentV1_AgentV1FunctionCallCancelled:
type: object
properties:
type:
type: string
enum:
- FunctionCallCancelled
description: Message type identifier for cancelled function calls
functions:
type: array
items:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1FunctionCallCancelledFunctionsItems'
description: The function calls that are no longer valid
required:
- type
- functions
title: AgentV1_AgentV1FunctionCallCancelled
AgentV1_AgentV1AgentStartedSpeaking:
type: object
properties:
type:
type: string
enum:
- AgentStartedSpeaking
description: Message type identifier for agent started speaking
total_latency:
type: string
title: float
description: Seconds from receiving the user's utterance to producing the agent's reply
tts_latency:
type: string
title: float
description: The portion of total latency attributable to text-to-speech
ttt_latency:
type: string
title: float
description: The portion of total latency attributable to text-to-text (usually an LLM)
required:
- type
- total_latency
- tts_latency
- ttt_latency
title: AgentV1_AgentV1AgentStartedSpeaking
AgentV1_AgentV1AgentAudioDone:
type: object
properties:
type:
type: string
enum:
- AgentAudioDone
description: Message type identifier indicating the agent has finished sending audio
required:
- type
title: AgentV1_AgentV1AgentAudioDone
ChannelsAgentV1MessagesAgentV1ErrorType:
type: string
enum:
- Error
description: Message type identifier for error responses
title: ChannelsAgentV1MessagesAgentV1ErrorType
AgentV1_AgentV1Error:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1ErrorType'
description: Message type identifier for error responses
description:
type: string
description: A description of what went wrong
code:
type: string
description: Error code identifying the type of error
required:
- type
- description
- code
title: AgentV1_AgentV1Error
ChannelsAgentV1MessagesAgentV1WarningType:
type: string
enum:
- Warning
description: Message type identifier for warnings
title: ChannelsAgentV1MessagesAgentV1WarningType
AgentV1_AgentV1Warning:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1WarningType'
description: Message type identifier for warnings
description:
type: string
description: Description of the warning
code:
type: string
description: Warning code identifier
required:
- type
- description
- code
description: Notifies the client of non-fatal errors or warnings
title: AgentV1_AgentV1Warning
ChannelsAgentV1MessagesAgentV1HistoryOneOf0Role:
type: string
enum:
- user
- assistant
description: Identifies who spoke the statement
title: ChannelsAgentV1MessagesAgentV1HistoryOneOf0Role
AgentV1AgentV1History0:
type: object
properties:
type:
type: string
enum:
- History
description: Message type identifier for conversation text
role:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1HistoryOneOf0Role'
description: Identifies who spoke the statement
content:
type: string
description: The actual statement that was spoken
required:
- type
- role
- content
description: Conversation text as part of the conversation history
title: AgentV1AgentV1History0
ChannelsAgentV1MessagesAgentV1HistoryOneOf1FunctionCallsItems:
type: object
properties:
id:
type: string
description: Unique identifier for the function call
name:
type: string
description: Name of the function called
client_side:
type: boolean
description: Indicates if the call was client-side or server-side
arguments:
type: string
description: Arguments passed to the function
response:
type: string
description: Response from the function call
thought_signature:
type: string
description: Some Gemini models require this as an additional function call identifier
required:
- id
- name
- client_side
- arguments
- response
title: ChannelsAgentV1MessagesAgentV1HistoryOneOf1FunctionCallsItems
AgentV1AgentV1History1:
type: object
properties:
type:
type: string
enum:
- History
function_calls:
type: array
items:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1HistoryOneOf1FunctionCallsItems'
description: List of function call objects
required:
- type
- function_calls
description: Client-side or server-side function call request and response as part of the conversation history
title: AgentV1AgentV1History1
AgentV1_AgentV1History:
oneOf:
- $ref: '#/components/schemas/AgentV1AgentV1History0'
- $ref: '#/components/schemas/AgentV1AgentV1History1'
description: A history message is either a conversational message or a function call
title: AgentV1_AgentV1History
AgentV1_AgentV1Audio:
type: string
format: binary
title: AgentV1_AgentV1Audio
ChannelsAgentV1MessagesAgentV1SettingsFlags:
type: object
properties:
history:
type: boolean
default: true
description: Enable or disable history message reporting
title: ChannelsAgentV1MessagesAgentV1SettingsFlags
ChannelsAgentV1MessagesAgentV1SettingsAudioInputEncoding:
type: string
enum:
- linear16
- linear32
- flac
- alaw
- mulaw
- amr-nb
- amr-wb
- opus
- ogg-opus
- speex
- g729
default: linear16
description: Audio encoding format
title: ChannelsAgentV1MessagesAgentV1SettingsAudioInputEncoding
ChannelsAgentV1MessagesAgentV1SettingsAudioInput:
type: object
properties:
encoding:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAudioInputEncoding'
description: Audio encoding format
sample_rate:
type: integer
default: 24000
description: Sample rate in Hz. Common values are 16000, 24000, 44100, 48000
required:
- encoding
- sample_rate
description: >-
Audio input configuration settings. If omitted, defaults to encoding=linear16 and sample_rate=24000. Higher
sample rates like 44100 Hz provide better audio quality.
title: ChannelsAgentV1MessagesAgentV1SettingsAudioInput
ChannelsAgentV1MessagesAgentV1SettingsAudioOutputEncoding:
type: string
enum:
- linear16
- mulaw
- alaw
- mp3
- opus
- flac
- aac
default: linear16
description: Audio encoding format for streaming TTS output
title: ChannelsAgentV1MessagesAgentV1SettingsAudioOutputEncoding
ChannelsAgentV1MessagesAgentV1SettingsAudioOutputContainer:
type: string
enum:
- none
- wav
- ogg
default: none
description: Audio container format.
title: ChannelsAgentV1MessagesAgentV1SettingsAudioOutputContainer
ChannelsAgentV1MessagesAgentV1SettingsAudioOutput:
type: object
properties:
encoding:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAudioOutputEncoding'
default: linear16
description: Audio encoding format for streaming TTS output
sample_rate:
type: integer
description: Sample rate in Hz
bitrate:
type: integer
description: Audio bitrate in bits per second
container:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAudioOutputContainer'
default: none
description: Audio container format.
description: Audio output configuration settings
title: ChannelsAgentV1MessagesAgentV1SettingsAudioOutput
ChannelsAgentV1MessagesAgentV1SettingsAudio:
type: object
properties:
input:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAudioInput'
description: >-
Audio input configuration settings. If omitted, defaults to encoding=linear16 and sample_rate=24000. Higher
sample rates like 44100 Hz provide better audio quality.
output:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAudioOutput'
description: Audio output configuration settings
title: ChannelsAgentV1MessagesAgentV1SettingsAudio
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItemsOneOf0Role:
type: string
enum:
- user
- assistant
description: Identifies who spoke the statement
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItemsOneOf0Role
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems0:
type: object
properties:
type:
type: string
enum:
- History
description: Message type identifier for conversation text
role:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItemsOneOf0Role'
description: Identifies who spoke the statement
content:
type: string
description: The actual statement that was spoken
required:
- type
- role
- content
description: Conversation text as part of the conversation history
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems0
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItemsOneOf1FunctionCallsItems:
type: object
properties:
id:
type: string
description: Unique identifier for the function call
name:
type: string
description: Name of the function called
client_side:
type: boolean
description: Indicates if the call was client-side or server-side
arguments:
type: string
description: Arguments passed to the function
response:
type: string
description: Response from the function call
thought_signature:
type: string
description: Some Gemini models require this as an additional function call identifier
required:
- id
- name
- client_side
- arguments
- response
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItemsOneOf1FunctionCallsItems
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems1:
type: object
properties:
type:
type: string
enum:
- History
function_calls:
type: array
items:
$ref: >-
#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItemsOneOf1FunctionCallsItems
description: List of function call objects
required:
- type
- function_calls
description: Client-side or server-side function call request and response as part of the conversation history
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems1
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems:
oneOf:
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems0'
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems1'
description: A history message is either a conversational message or a function call
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Context:
type: object
properties:
messages:
type: array
items:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ContextMessagesItems'
description: Conversation history as a list of messages and function calls
description: Conversation context including the history of messages and function calls
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Context
DeepgramListenProviderV1:
type: object
properties:
type:
type: string
enum:
- deepgram
description: Provider type for speech-to-text
version:
type: string
enum:
- v1
description: Specifies usage of the V1 Deepgram speech-to-text API
model:
type: string
description: Model to use for speech to text using the V1 API (e.g. Nova-3, Nova-2)
language:
type: string
default: en-US
description: >-
Language code to use for speech-to-text. Can be a BCP-47 language tag (e.g. `en`), or `multi` for
code-switching transcription
keyterms:
type: array
items:
type: string
description: Prompt keyterm recognition to improve Keyword Recall Rate
smart_format:
type: boolean
default: false
description: Applies smart formatting to improve transcript readability
required:
- type
title: DeepgramListenProviderV1
DeepgramListenProviderV2:
type: object
properties:
type:
type: string
enum:
- deepgram
description: Provider type for speech-to-text
version:
type: string
enum:
- v2
description: Specifies usage of the V2 Deepgram speech-to-text API (e.g. Flux)
model:
type: string
description: Model to use for speech to text using the V2 API (e.g. flux-general-en, flux-general-multi)
language_hints:
type: array
items:
type: string
description: >-
An array of one or more BCP-47 language codes to bias the model toward specific languages. Only supported
when model is flux-general-multi. Without hints, the model auto-detects the spoken language. See the
Language Prompting guide for details.
eot_threshold:
type: number
format: double
description: >-
End-of-turn confidence required to finish a turn. Valid range: 0.5 - 1.0. Defaults to 0.7. Set to 1.0 to
fully suppress confidence-based end-of-turn detection. `eot_timeout_ms` still ends idle turns; increase it
when using ForceEndTurn for full manual turn control.
eager_eot_threshold:
type: number
format: double
description: >-
End-of-turn confidence required to fire an eager end-of-turn event. When set, enables EagerEndOfTurn and
TurnResumed events. Valid range: 0.3 - 0.9.
eot_timeout_ms:
type: integer
description: >-
A turn will be finished when this much time in milliseconds has passed after speech, regardless of EOT
confidence. Defaults to 5000.
keyterms:
type: array
items:
type: string
description: Prompt keyterm recognition to improve Keyword Recall Rate
required:
- type
- model
title: DeepgramListenProviderV2
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ListenProvider:
oneOf:
- $ref: '#/components/schemas/DeepgramListenProviderV1'
- $ref: '#/components/schemas/DeepgramListenProviderV2'
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ListenProvider
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Listen:
type: object
properties:
provider:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0ListenProvider'
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Listen
OpenAiThinkProviderVersion:
type: string
enum:
- v1
description: The REST API version for the OpenAI chat completions API
title: OpenAiThinkProviderVersion
OpenAiThinkProviderModel:
type: string
enum:
- gpt-5
- gpt-5-mini
- gpt-5-nano
- gpt-4.1
- gpt-4.1-mini
- gpt-4.1-nano
- gpt-4o
- gpt-4o-mini
description: OpenAI model to use
title: OpenAiThinkProviderModel
OpenAiThinkProviderReasoningMode:
type: string
enum:
- none
- minimal
- low
- medium
- high
description: OpenAI reasoning_effort
title: OpenAiThinkProviderReasoningMode
OpenAiThinkProvider:
type: object
properties:
type:
type: string
enum:
- open_ai
version:
$ref: '#/components/schemas/OpenAiThinkProviderVersion'
description: The REST API version for the OpenAI chat completions API
model:
$ref: '#/components/schemas/OpenAiThinkProviderModel'
description: OpenAI model to use
temperature:
type: number
format: double
minimum: 0
maximum: 2
description: OpenAI temperature (0-2)
reasoning_mode:
$ref: '#/components/schemas/OpenAiThinkProviderReasoningMode'
description: OpenAI reasoning_effort
required:
- type
- model
title: OpenAiThinkProvider
AwsBedrockThinkProviderModel:
type: string
enum:
- anthropic/claude-3-5-sonnet-20240620-v1:0
- anthropic/claude-3-5-haiku-20240307-v1:0
description: AWS Bedrock model to use
title: AwsBedrockThinkProviderModel
AwsBedrockThinkProviderCredentialsType:
type: string
enum:
- sts
- iam
description: AWS credentials type (STS short-lived or IAM long-lived)
title: AwsBedrockThinkProviderCredentialsType
AwsBedrockThinkProviderCredentials:
type: object
properties:
type:
$ref: '#/components/schemas/AwsBedrockThinkProviderCredentialsType'
description: AWS credentials type (STS short-lived or IAM long-lived)
region:
type: string
description: AWS region
access_key_id:
type: string
description: AWS access key
secret_access_key:
type: string
description: AWS secret access key
session_token:
type: string
description: AWS session token (required for STS only)
description: AWS credentials type (STS short-lived or IAM long-lived)
title: AwsBedrockThinkProviderCredentials
AwsBedrockThinkProvider:
type: object
properties:
type:
type: string
enum:
- aws_bedrock
model:
$ref: '#/components/schemas/AwsBedrockThinkProviderModel'
description: AWS Bedrock model to use
temperature:
type: number
format: double
minimum: 0
maximum: 2
description: AWS Bedrock temperature (0-2)
credentials:
$ref: '#/components/schemas/AwsBedrockThinkProviderCredentials'
description: AWS credentials type (STS short-lived or IAM long-lived)
required:
- type
- model
title: AwsBedrockThinkProvider
AnthropicThinkProviderVersion:
type: string
enum:
- v1
description: The REST API version for the Anthropic Messages API
title: AnthropicThinkProviderVersion
AnthropicThinkProviderModel:
type: string
enum:
- claude-3-5-haiku-latest
- claude-sonnet-4-20250514
description: Anthropic model to use
title: AnthropicThinkProviderModel
AnthropicThinkProvider:
type: object
properties:
type:
type: string
enum:
- anthropic
version:
$ref: '#/components/schemas/AnthropicThinkProviderVersion'
description: The REST API version for the Anthropic Messages API
model:
$ref: '#/components/schemas/AnthropicThinkProviderModel'
description: Anthropic model to use
temperature:
type: number
format: double
minimum: 0
maximum: 1
description: Anthropic temperature (0-1)
required:
- type
- model
title: AnthropicThinkProvider
GoogleThinkProviderVersion:
type: string
enum:
- ai-studio-v1beta
- gemini-enterprise-agent-v1
- v1beta
description: >-
The Google API used for the request: ai-studio-v1beta for the AI Studio API, or gemini-enterprise-agent-v1 for
the Gemini Enterprise Agent (GEA) API. v1beta is accepted as an alias for ai-studio-v1beta. Defaults based on
the Deepgram Voice Agent endpoint you connect to.
title: GoogleThinkProviderVersion
GoogleThinkProviderModel:
type: string
enum:
- gemini-2.0-flash
- gemini-2.0-flash-lite
- gemini-2.5-flash
description: Google model to use
title: GoogleThinkProviderModel
GoogleThinkProvider:
type: object
properties:
type:
type: string
enum:
- google
version:
$ref: '#/components/schemas/GoogleThinkProviderVersion'
description: >-
The Google API used for the request: ai-studio-v1beta for the AI Studio API, or gemini-enterprise-agent-v1
for the Gemini Enterprise Agent (GEA) API. v1beta is accepted as an alias for ai-studio-v1beta. Defaults
based on the Deepgram Voice Agent endpoint you connect to.
model:
$ref: '#/components/schemas/GoogleThinkProviderModel'
description: Google model to use
temperature:
type: number
format: double
minimum: 0
maximum: 2
description: Google temperature (0-2)
required:
- type
- model
title: GoogleThinkProvider
GroqThinkProviderVersion:
type: string
enum:
- v1
description: The REST API version for the Groq's chat completions API (mostly OpenAI-compatible)
title: GroqThinkProviderVersion
GroqThinkProviderModel:
type: string
enum:
- openai/gpt-oss-20b
description: Groq model to use
title: GroqThinkProviderModel
GroqThinkProviderReasoningMode:
type: string
enum:
- none
- minimal
- low
- medium
- high
description: OpenAI reasoning_effort
title: GroqThinkProviderReasoningMode
GroqThinkProvider:
type: object
properties:
type:
type: string
enum:
- groq
version:
$ref: '#/components/schemas/GroqThinkProviderVersion'
description: The REST API version for the Groq's chat completions API (mostly OpenAI-compatible)
model:
$ref: '#/components/schemas/GroqThinkProviderModel'
description: Groq model to use
temperature:
type: number
format: double
minimum: 0
maximum: 2
description: Groq temperature (0-2)
reasoning_mode:
$ref: '#/components/schemas/GroqThinkProviderReasoningMode'
description: OpenAI reasoning_effort
required:
- type
- model
title: GroqThinkProvider
ThinkSettingsV1Provider:
oneOf:
- $ref: '#/components/schemas/OpenAiThinkProvider'
- $ref: '#/components/schemas/AwsBedrockThinkProvider'
- $ref: '#/components/schemas/AnthropicThinkProvider'
- $ref: '#/components/schemas/GoogleThinkProvider'
- $ref: '#/components/schemas/GroqThinkProvider'
title: ThinkSettingsV1Provider
ThinkSettingsV1Endpoint:
type: object
properties:
url:
type: string
description: Custom LLM endpoint URL
headers:
type: object
additionalProperties:
type: string
description: Custom headers for the endpoint
description: |
Optional for non-Deepgram LLM providers. When present, must include url field and headers object
title: ThinkSettingsV1Endpoint
ThinkSettingsV1FunctionsItemsParameters:
type: object
properties:
description: Function parameters
title: ThinkSettingsV1FunctionsItemsParameters
ThinkSettingsV1FunctionsItemsEndpoint:
type: object
properties:
url:
type: string
description: Endpoint URL
method:
type: string
description: HTTP method
headers:
type: object
additionalProperties:
type: string
description: The Function endpoint to call. if not passed, function is called client-side
title: ThinkSettingsV1FunctionsItemsEndpoint
ThinkSettingsV1FunctionsItems:
type: object
properties:
name:
type: string
description: Function name
description:
type: string
description: Function description
parameters:
$ref: '#/components/schemas/ThinkSettingsV1FunctionsItemsParameters'
description: Function parameters
defer_until_eot:
type: boolean
default: false
description: >-
Hold this function call until the user's turn is confirmed instead of dispatching it speculatively. Set it
to true for actions that cannot be undone. If the turn resumes, a deferred call is discarded before it runs.
Defaults to false
endpoint:
$ref: '#/components/schemas/ThinkSettingsV1FunctionsItemsEndpoint'
description: The Function endpoint to call. if not passed, function is called client-side
title: ThinkSettingsV1FunctionsItems
ThinkSettingsV1ContextLength0:
type: string
enum:
- max
description: Agent will not discard context regardless of length
title: ThinkSettingsV1ContextLength0
ThinkSettingsV1ContextLength:
oneOf:
- $ref: '#/components/schemas/ThinkSettingsV1ContextLength0'
- type: number
format: double
minimum: 2
description: >
Specifies the number of characters retained in context between user messages, agent responses, and function
calls. This setting is only configurable when a custom think endpoint is used
title: ThinkSettingsV1ContextLength
ThinkSettingsV1:
type: object
properties:
provider:
$ref: '#/components/schemas/ThinkSettingsV1Provider'
endpoint:
$ref: '#/components/schemas/ThinkSettingsV1Endpoint'
description: |
Optional for non-Deepgram LLM providers. When present, must include url field and headers object
functions:
type: array
items:
$ref: '#/components/schemas/ThinkSettingsV1FunctionsItems'
prompt:
type: string
context_length:
$ref: '#/components/schemas/ThinkSettingsV1ContextLength'
description: >
Specifies the number of characters retained in context between user messages, agent responses, and function
calls. This setting is only configurable when a custom think endpoint is used
required:
- provider
title: ThinkSettingsV1
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Think1:
type: array
items:
$ref: '#/components/schemas/ThinkSettingsV1'
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Think1
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Think:
oneOf:
- $ref: '#/components/schemas/ThinkSettingsV1'
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Think1'
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Think
DeepgramSpeakProviderModel:
type: string
enum:
- aura-asteria-en
- aura-luna-en
- aura-stella-en
- aura-athena-en
- aura-hera-en
- aura-orion-en
- aura-arcas-en
- aura-perseus-en
- aura-angus-en
- aura-orpheus-en
- aura-helios-en
- aura-zeus-en
- aura-2-amalthea-en
- aura-2-andromeda-en
- aura-2-apollo-en
- aura-2-arcas-en
- aura-2-aries-en
- aura-2-asteria-en
- aura-2-athena-en
- aura-2-atlas-en
- aura-2-aurora-en
- aura-2-callista-en
- aura-2-cora-en
- aura-2-cordelia-en
- aura-2-delia-en
- aura-2-draco-en
- aura-2-electra-en
- aura-2-harmonia-en
- aura-2-helena-en
- aura-2-hera-en
- aura-2-hermes-en
- aura-2-hyperion-en
- aura-2-iris-en
- aura-2-janus-en
- aura-2-juno-en
- aura-2-jupiter-en
- aura-2-luna-en
- aura-2-mars-en
- aura-2-minerva-en
- aura-2-neptune-en
- aura-2-odysseus-en
- aura-2-ophelia-en
- aura-2-orion-en
- aura-2-orpheus-en
- aura-2-pandora-en
- aura-2-phoebe-en
- aura-2-pluto-en
- aura-2-saturn-en
- aura-2-selene-en
- aura-2-thalia-en
- aura-2-theia-en
- aura-2-vesta-en
- aura-2-zeus-en
- aura-2-sirio-es
- aura-2-nestor-es
- aura-2-carina-es
- aura-2-celeste-es
- aura-2-alvaro-es
- aura-2-diana-es
- aura-2-aquila-es
- aura-2-selena-es
- aura-2-estrella-es
- aura-2-javier-es
- flux-alexis-en
- flux-bree-en
- flux-brittany-en
- flux-brooke-en
- flux-bruce-en
- flux-cliff-en
- flux-cole-en
- flux-colin-en
- flux-conor-en
- flux-donovan-en
- flux-drew-en
- flux-elise-en
- flux-gemma-en
- flux-haley-en
- flux-hannah-en
- flux-heather-en
- flux-jack-en
- flux-kai-en
- flux-kelsey-en
- flux-kit-en
- flux-maeve-en
- flux-marcelo-en
- flux-marcus-en
- flux-meena-en
- flux-meghan-en
- flux-miles-en
- flux-naveen-en
- flux-paige-en
- flux-priya-en
- flux-rufus-en
- flux-sean-en
- flux-sharon-en
- flux-sienna-en
- flux-tanner-en
- flux-wade-en
- flux-wes-en
description: >-
Deepgram TTS model. Aura models (version v1) use the aura-* voices; Flux TTS (version v2) uses the
flux-- voices (e.g. flux-alexis-en). Defaults to flux-kit-en when agent.speak is omitted.
title: DeepgramSpeakProviderModel
DeepgramSpeakProviderExpressivity:
type: string
enum:
- '-2'
- '-1'
- '0'
- '1'
- '2'
description: >-
Delivery register of the generated speech, on a calm-to-animated axis. Flux TTS (version v2) only, on every Flux
voice. Accepts the whole numbers -2 to 2, where 0 (the default) is the voice's tuned delivery and the only value
validated for production, -2 the calm end of the range and 2 the animated end. Fixed for the session. Beta:
behavior may change in future model versions, and non-default values increase the risk of hallucinations and
pronunciation errors. See [Expressivity](/guides/tts-voice-controls-tts-expressivity).
title: DeepgramSpeakProviderExpressivity
DeepgramSpeakProvider:
type: object
properties:
type:
type: string
enum:
- deepgram
version:
type: string
default: v1
description: >-
The Deepgram text-to-speech model family. Accepted values: `v1` (Aura, the default) and `v2` (Flux TTS). Use
`v1` with an aura-* model and `v2` with a flux-* model. Defaults to `v1` when omitted.
model:
$ref: '#/components/schemas/DeepgramSpeakProviderModel'
description: >-
Deepgram TTS model. Aura models (version v1) use the aura-* voices; Flux TTS (version v2) uses the
flux-- voices (e.g. flux-alexis-en). Defaults to flux-kit-en when agent.speak is omitted.
speed:
type: number
format: double
minimum: 0.5
maximum: 1.5
default: 1
description: >-
Speaking rate multiplier that adjusts the pace of generated speech while preserving natural prosody and
voice quality. Aura (version v1) accepts any value from 0.7 to 1.5. Flux TTS (version v2) accepts values
from 0.5 to 1.5 in 0.05 increments; a value the family does not accept ends the session with
FAILED_TO_SPEAK. Not yet supported in all languages.
expressivity:
$ref: '#/components/schemas/DeepgramSpeakProviderExpressivity'
default: 0
description: >-
Delivery register of the generated speech, on a calm-to-animated axis. Flux TTS (version v2) only, on every
Flux voice. Accepts the whole numbers -2 to 2, where 0 (the default) is the voice's tuned delivery and the
only value validated for production, -2 the calm end of the range and 2 the animated end. Fixed for the
session. Beta: behavior may change in future model versions, and non-default values increase the risk of
hallucinations and pronunciation errors. See [Expressivity](/guides/tts-voice-controls-tts-expressivity).
required:
- type
- model
description: >-
Deepgram text-to-speech provider. Aura models use version v1 (default); Flux TTS uses version v2 and a flux-*
model. Flux TTS is the default when agent.speak is omitted, using the flux-kit-en voice.
title: DeepgramSpeakProvider
ElevenLabsSpeakProviderVersion:
type: string
enum:
- v1
description: The REST API version for the ElevenLabs text-to-speech API
title: ElevenLabsSpeakProviderVersion
ElevenLabsSpeakProviderModelId:
type: string
enum:
- eleven_turbo_v2_5
- eleven_monolingual_v1
- eleven_multilingual_v2
description: Eleven Labs model ID
title: ElevenLabsSpeakProviderModelId
ElevenLabsSpeakProvider:
type: object
properties:
type:
type: string
enum:
- eleven_labs
version:
$ref: '#/components/schemas/ElevenLabsSpeakProviderVersion'
description: The REST API version for the ElevenLabs text-to-speech API
model_id:
$ref: '#/components/schemas/ElevenLabsSpeakProviderModelId'
description: Eleven Labs model ID
language:
type: string
description: Optional language to use, e.g. 'en-US'. Corresponds to the `language_code` parameter in the ElevenLabs API
language_code:
type: string
description: Use the `language` field instead.
deprecated: true
required:
- type
- model_id
title: ElevenLabsSpeakProvider
CartesiaSpeakProviderVersion:
type: string
enum:
- '2025-03-17'
description: The API version header for the Cartesia text-to-speech API
title: CartesiaSpeakProviderVersion
CartesiaSpeakProviderModelId:
type: string
enum:
- sonic-2
- sonic-multilingual
description: Cartesia model ID
title: CartesiaSpeakProviderModelId
CartesiaSpeakProviderVoice:
type: object
properties:
mode:
type: string
description: Cartesia voice mode
id:
type: string
description: Cartesia voice ID
required:
- mode
- id
title: CartesiaSpeakProviderVoice
CartesiaSpeakProvider:
type: object
properties:
type:
type: string
enum:
- cartesia
version:
$ref: '#/components/schemas/CartesiaSpeakProviderVersion'
description: The API version header for the Cartesia text-to-speech API
model_id:
$ref: '#/components/schemas/CartesiaSpeakProviderModelId'
description: Cartesia model ID
voice:
$ref: '#/components/schemas/CartesiaSpeakProviderVoice'
language:
type: string
description: Cartesia language code
volume:
type: number
format: double
minimum: 0.5
maximum: 2
description: >
Volume level for Cartesia TTS output. Valid range: 0.5 to 2.0. See [Cartesia
documentation](https://docs.cartesia.ai/build-with-cartesia/sonic-3/volume-speed-emotion#volume-speed-and-emotion).
required:
- type
- model_id
- voice
title: CartesiaSpeakProvider
OpenAiSpeakProviderVersion:
type: string
enum:
- v1
description: The REST API version for the OpenAI text-to-speech API
title: OpenAiSpeakProviderVersion
OpenAiSpeakProviderModel:
type: string
enum:
- tts-1
- tts-1-hd
description: OpenAI TTS model
title: OpenAiSpeakProviderModel
OpenAiSpeakProviderVoice:
type: string
enum:
- alloy
- echo
- fable
- onyx
- nova
- shimmer
description: OpenAI voice
title: OpenAiSpeakProviderVoice
OpenAiSpeakProvider:
type: object
properties:
type:
type: string
enum:
- open_ai
version:
$ref: '#/components/schemas/OpenAiSpeakProviderVersion'
description: The REST API version for the OpenAI text-to-speech API
model:
$ref: '#/components/schemas/OpenAiSpeakProviderModel'
description: OpenAI TTS model
voice:
$ref: '#/components/schemas/OpenAiSpeakProviderVoice'
description: OpenAI voice
required:
- type
- model
- voice
title: OpenAiSpeakProvider
AwsPollySpeakProviderVoice:
type: string
enum:
- Matthew
- Joanna
- Amy
- Emma
- Brian
- Arthur
- Aria
- Ayanda
description: AWS Polly voice name
title: AwsPollySpeakProviderVoice
AwsPollySpeakProviderEngine:
type: string
enum:
- generative
- long-form
- standard
- neural
title: AwsPollySpeakProviderEngine
AwsPollySpeakProviderCredentialsType:
type: string
enum:
- sts
- iam
title: AwsPollySpeakProviderCredentialsType
AwsPollySpeakProviderCredentials:
type: object
properties:
type:
$ref: '#/components/schemas/AwsPollySpeakProviderCredentialsType'
region:
type: string
access_key_id:
type: string
secret_access_key:
type: string
session_token:
type: string
description: Required for STS only
required:
- type
- region
- access_key_id
- secret_access_key
title: AwsPollySpeakProviderCredentials
AwsPollySpeakProvider:
type: object
properties:
type:
type: string
enum:
- aws_polly
voice:
$ref: '#/components/schemas/AwsPollySpeakProviderVoice'
description: AWS Polly voice name
language:
type: string
description: Language code to use, e.g. 'en-US'. Corresponds to the `language_code` parameter in the AWS Polly API
language_code:
type: string
description: Use the `language` field instead.
deprecated: true
engine:
$ref: '#/components/schemas/AwsPollySpeakProviderEngine'
credentials:
$ref: '#/components/schemas/AwsPollySpeakProviderCredentials'
required:
- type
- voice
- language
- engine
- credentials
title: AwsPollySpeakProvider
SpeakSettingsV1Provider:
oneOf:
- $ref: '#/components/schemas/DeepgramSpeakProvider'
- $ref: '#/components/schemas/ElevenLabsSpeakProvider'
- $ref: '#/components/schemas/CartesiaSpeakProvider'
- $ref: '#/components/schemas/OpenAiSpeakProvider'
- $ref: '#/components/schemas/AwsPollySpeakProvider'
title: SpeakSettingsV1Provider
SpeakSettingsV1Endpoint:
type: object
properties:
url:
type: string
description: >
Custom TTS endpoint URL. Cannot contain `output_format` or `model_id` query parameters when the provider is
Eleven Labs.
headers:
type: object
additionalProperties:
type: string
description: >
Optional if provider is Deepgram. Required for non-Deepgram TTS providers.
When present, must include url field and headers object. Valid schemes are https and wss with wss only supported
for Eleven Labs.
title: SpeakSettingsV1Endpoint
SpeakSettingsV1:
type: object
properties:
provider:
$ref: '#/components/schemas/SpeakSettingsV1Provider'
endpoint:
$ref: '#/components/schemas/SpeakSettingsV1Endpoint'
description: >
Optional if provider is Deepgram. Required for non-Deepgram TTS providers.
When present, must include url field and headers object. Valid schemes are https and wss with wss only
supported for Eleven Labs.
required:
- provider
title: SpeakSettingsV1
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Speak1:
type: array
items:
$ref: '#/components/schemas/SpeakSettingsV1'
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Speak1
ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Speak:
oneOf:
- $ref: '#/components/schemas/SpeakSettingsV1'
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Speak1'
title: ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Speak
ChannelsAgentV1MessagesAgentV1SettingsAgent0:
type: object
properties:
language:
type: string
default: en
description: Deprecated. Use `listen.provider.language` and `speak.provider.language` fields instead.
deprecated: true
context:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Context'
description: Conversation context including the history of messages and function calls
listen:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Listen'
think:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Think'
speak:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgentOneOf0Speak'
greeting:
type: string
description: Optional message that agent will speak at the start
title: ChannelsAgentV1MessagesAgentV1SettingsAgent0
ChannelsAgentV1MessagesAgentV1SettingsAgent:
oneOf:
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgent0'
- type: string
format: uuid
title: ChannelsAgentV1MessagesAgentV1SettingsAgent
AgentV1_AgentV1Settings:
type: object
properties:
type:
type: string
enum:
- Settings
tags:
type: array
items:
type: string
description: Tags to associate with the request
experimental:
type: boolean
default: false
description: To enable experimental features
flags:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsFlags'
mip_opt_out:
type: boolean
default: false
description: To opt out of Deepgram Model Improvement Program
audio:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAudio'
agent:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1SettingsAgent'
required:
- type
- audio
- agent
title: AgentV1_AgentV1Settings
ChannelsAgentV1MessagesAgentV1UpdateListenListenProvider:
oneOf:
- $ref: '#/components/schemas/DeepgramListenProviderV1'
- $ref: '#/components/schemas/DeepgramListenProviderV2'
title: ChannelsAgentV1MessagesAgentV1UpdateListenListenProvider
ChannelsAgentV1MessagesAgentV1UpdateListenListen:
type: object
properties:
provider:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1UpdateListenListenProvider'
required:
- provider
description: >-
Listen configuration to update. Contains a provider object with the same schema as Settings. The model and
language can be changed mid-session. Keyterms can only be updated mid-session for Flux models.
title: ChannelsAgentV1MessagesAgentV1UpdateListenListen
AgentV1_AgentV1UpdateListen:
type: object
properties:
type:
type: string
enum:
- UpdateListen
description: Message type identifier for updating the listen configuration
listen:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1UpdateListenListen'
description: >-
Listen configuration to update. Contains a provider object with the same schema as Settings. The model and
language can be changed mid-session. Keyterms can only be updated mid-session for Flux models.
required:
- type
- listen
title: AgentV1_AgentV1UpdateListen
ChannelsAgentV1MessagesAgentV1UpdateThinkThink1:
type: array
items:
$ref: '#/components/schemas/ThinkSettingsV1'
title: ChannelsAgentV1MessagesAgentV1UpdateThinkThink1
ChannelsAgentV1MessagesAgentV1UpdateThinkThink:
oneOf:
- $ref: '#/components/schemas/ThinkSettingsV1'
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1UpdateThinkThink1'
title: ChannelsAgentV1MessagesAgentV1UpdateThinkThink
AgentV1_AgentV1UpdateThink:
type: object
properties:
type:
type: string
enum:
- UpdateThink
description: Message type identifier for updating the think model
think:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1UpdateThinkThink'
required:
- type
- think
title: AgentV1_AgentV1UpdateThink
ChannelsAgentV1MessagesAgentV1UpdateSpeakSpeak1:
type: array
items:
$ref: '#/components/schemas/SpeakSettingsV1'
title: ChannelsAgentV1MessagesAgentV1UpdateSpeakSpeak1
ChannelsAgentV1MessagesAgentV1UpdateSpeakSpeak:
oneOf:
- $ref: '#/components/schemas/SpeakSettingsV1'
- $ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1UpdateSpeakSpeak1'
title: ChannelsAgentV1MessagesAgentV1UpdateSpeakSpeak
AgentV1_AgentV1UpdateSpeak:
type: object
properties:
type:
type: string
enum:
- UpdateSpeak
description: Message type identifier for updating the speak model
speak:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1UpdateSpeakSpeak'
required:
- type
- speak
title: AgentV1_AgentV1UpdateSpeak
AgentV1_AgentV1InjectUserMessage:
type: object
properties:
type:
type: string
enum:
- InjectUserMessage
description: Message type identifier for injecting a user message
content:
type: string
description: The specific phrase or statement the agent should respond to
required:
- type
- content
title: AgentV1_AgentV1InjectUserMessage
ChannelsAgentV1MessagesAgentV1InjectAgentMessageBehavior:
type: string
enum:
- default
- queue
- interrupt
default: default
description: >
Controls how the injection interacts with any in-progress user or agent turn.
* `default` — The agent speaks only if neither the user nor the agent is mid-turn. If a turn is in progress, the
server replies with `InjectionRefused`.
* `queue` — The message is appended after any already-queued `ConversationText` without interrupting the current
agent turn or think response. If nothing is queued, the message plays immediately.
* `interrupt` — The agent immediately speaks. If the agent was already speaking, it interrupts the current
speech and replaces it with the new message. If the user is speaking, the agent interrupts with the new message,
but the user's continued speech triggers `UserStartedSpeaking`, which quickly interrupts the agent.
title: ChannelsAgentV1MessagesAgentV1InjectAgentMessageBehavior
AgentV1_AgentV1InjectAgentMessage:
type: object
properties:
type:
type: string
enum:
- InjectAgentMessage
description: Message type identifier for injecting an agent message
message:
type: string
description: The statement that the agent should say
behavior:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1InjectAgentMessageBehavior'
default: default
description: >
Controls how the injection interacts with any in-progress user or agent turn.
* `default` — The agent speaks only if neither the user nor the agent is mid-turn. If a turn is in progress,
the server replies with `InjectionRefused`.
* `queue` — The message is appended after any already-queued `ConversationText` without interrupting the
current agent turn or think response. If nothing is queued, the message plays immediately.
* `interrupt` — The agent immediately speaks. If the agent was already speaking, it interrupts the current
speech and replaces it with the new message. If the user is speaking, the agent interrupts with the new
message, but the user's continued speech triggers `UserStartedSpeaking`, which quickly interrupts the agent.
required:
- type
- message
title: AgentV1_AgentV1InjectAgentMessage
AgentV1_AgentV1SendFunctionCallResponse:
type: object
properties:
type:
type: string
enum:
- FunctionCallResponse
description: Message type identifier for function call responses
id:
type: string
description: |
The unique identifier for the function call.
• **Required for client responses**: Should match the id from
the corresponding `FunctionCallRequest`
• **Optional for server responses**: Server may omit when responding
to internal function executions
name:
type: string
description: The name of the function being called
content:
type: string
description: The content or result of the function call
required:
- type
- name
- content
description: |
Function call response message used bidirectionally:
• **Client → Server**: Response after client executes a function
marked as client_side: true
• **Server → Client**: Response after server executes a function
marked as client_side: false
The same message structure serves both directions, enabling a unified
interface for function call responses regardless of execution location.
title: AgentV1_AgentV1SendFunctionCallResponse
ChannelsAgentV1MessagesAgentV1KeepAliveType:
type: string
enum:
- KeepAlive
description: Message type identifier
title: ChannelsAgentV1MessagesAgentV1KeepAliveType
AgentV1_AgentV1KeepAlive:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsAgentV1MessagesAgentV1KeepAliveType'
description: Message type identifier
required:
- type
description: Send a control message to the agent
title: AgentV1_AgentV1KeepAlive
AgentV1_AgentV1UpdatePrompt:
type: object
properties:
type:
type: string
enum:
- UpdatePrompt
description: Message type identifier for prompt update request
prompt:
type: string
description: The new system prompt to be used by the agent
required:
- type
- prompt
title: AgentV1_AgentV1UpdatePrompt
AgentV1_AgentV1ForceEndTurn:
type: object
properties:
type:
type: string
enum:
- ForceEndTurn
description: Message type identifier for forcing the end of the current turn
required:
- type
title: AgentV1_AgentV1ForceEndTurn
AgentV1_AgentV1Media:
type: string
format: binary
title: AgentV1_AgentV1Media
ListenV1Callback:
description: Any type
title: ListenV1Callback
ListenV1CallbackMethod:
type: string
enum:
- POST
- GET
- PUT
- DELETE
default: POST
description: HTTP method by which the callback request will be made
title: ListenV1CallbackMethod
ListenV1Channels:
description: Any type
title: ListenV1Channels
ListenV1DetectEntities:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: >-
Identifies and extracts key entities from content in submitted audio. Entities appear in final results. When
enabled, Punctuation will also be enabled by default
title: ListenV1DetectEntities
ListenV1Diarize:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: >-
Deprecated. Use `diarize_model` instead. Defaults to `false`. Recognize speaker changes. Each word in the
transcript will be assigned a speaker number starting at 0
title: ListenV1Diarize
deprecated: true
ListenV1_diarize_model:
type: string
enum:
- latest
- v1
title: ListenV1_diarize_model
ListenV1Dictation:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: Identify and extract key entities from content in submitted audio
title: ListenV1Dictation
ListenV1Encoding:
type: string
enum:
- linear16
- linear32
- flac
- alaw
- mulaw
- amr-nb
- amr-wb
- opus
- ogg-opus
- speex
- g729
description: Specify the expected encoding of your submitted audio
title: ListenV1Encoding
ListenV1Endpointing:
description: Any type
title: ListenV1Endpointing
ListenV1Extra:
description: Any type
title: ListenV1Extra
ListenV1InterimResults:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: >-
Specifies whether the streaming endpoint should provide ongoing transcription updates as more audio is received.
When set to true, the endpoint sends continuous updates, meaning transcription results may evolve over time
title: ListenV1InterimResults
ListenV1Keyterm:
description: Any type
title: ListenV1Keyterm
ListenV1Keywords:
description: Any type
title: ListenV1Keywords
ListenV1Language:
description: Any type
title: ListenV1Language
ListenV1MipOptOut:
description: Any type
title: ListenV1MipOptOut
ListenV1Model:
type: string
enum:
- nova-3
- nova-3-general
- nova-3-medical
- nova-2
- nova-2-general
- nova-2-meeting
- nova-2-finance
- nova-2-conversationalai
- nova-2-voicemail
- nova-2-video
- nova-2-medical
- nova-2-drivethru
- nova-2-automotive
- nova
- nova-general
- nova-phonecall
- nova-medical
- enhanced
- enhanced-general
- enhanced-meeting
- enhanced-phonecall
- enhanced-finance
- base
- meeting
- phonecall
- finance
- conversationalai
- voicemail
- video
- custom
description: AI model to use for the transcription
title: ListenV1Model
ListenV1Multichannel:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: Transcribe each audio channel independently
title: ListenV1Multichannel
ListenV1Numerals:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: Convert numbers from written format to numerical format
title: ListenV1Numerals
ListenV1ProfanityFilter:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: >-
Profanity Filter looks for recognized profanity and converts it to the nearest recognized non-profane word or
removes it from the transcript completely
title: ListenV1ProfanityFilter
ListenV1Punctuate:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: Add punctuation and capitalization to the transcript
title: ListenV1Punctuate
ListenV1Redact:
type: string
enum:
- 'true'
- 'false'
- pci
- numbers
- aggressive_numbers
- ssn
default: 'false'
description: Redaction removes sensitive information from your transcripts
title: ListenV1Redact
ListenV1Replace:
description: Any type
title: ListenV1Replace
ListenV1SampleRate:
description: Any type
title: ListenV1SampleRate
ListenV1Search:
description: Any type
title: ListenV1Search
ListenV1SmartFormat:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: >-
Apply formatting to transcript output. When set to true, additional formatting will be applied to transcripts to
improve readability
title: ListenV1SmartFormat
ListenV1Tag:
description: Any type
title: ListenV1Tag
ListenV1UtteranceEndMs:
description: Any type
title: ListenV1UtteranceEndMs
ListenV1VadEvents:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: Indicates that speech has started. You'll begin receiving Speech Started messages upon speech starting
title: ListenV1VadEvents
ListenV1Version:
description: Any type
title: ListenV1Version
ChannelsListenV1MessagesListenV1ResultsType:
type: string
enum:
- Results
description: Message type identifier
title: ChannelsListenV1MessagesListenV1ResultsType
ChannelsListenV1MessagesListenV1ResultsChannelAlternativesItemsWordsItems:
type: object
properties:
word:
type: string
description: The word of the transcription
start:
type: number
format: double
description: The start time of the word
end:
type: number
format: double
description: The end time of the word
confidence:
type: number
format: double
description: The confidence of the word
language:
type: string
description: The language of the word
punctuated_word:
type: string
description: The punctuated word of the word
speaker:
type: integer
description: The speaker of the word, present when diarization is enabled
required:
- word
- start
- end
- confidence
title: ChannelsListenV1MessagesListenV1ResultsChannelAlternativesItemsWordsItems
ChannelsListenV1MessagesListenV1ResultsChannelAlternativesItems:
type: object
properties:
transcript:
type: string
description: The transcript of the transcription
confidence:
type: number
format: double
description: The confidence of the transcription
languages:
type: array
items:
type: string
words:
type: array
items:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsChannelAlternativesItemsWordsItems'
required:
- transcript
- confidence
- words
title: ChannelsListenV1MessagesListenV1ResultsChannelAlternativesItems
ChannelsListenV1MessagesListenV1ResultsChannel:
type: object
properties:
alternatives:
type: array
items:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsChannelAlternativesItems'
required:
- alternatives
title: ChannelsListenV1MessagesListenV1ResultsChannel
ChannelsListenV1MessagesListenV1ResultsMetadataModelInfo:
type: object
properties:
name:
type: string
description: The name of the model
version:
type: string
description: The version of the model
arch:
type: string
description: The arch of the model
required:
- name
- version
- arch
title: ChannelsListenV1MessagesListenV1ResultsMetadataModelInfo
ChannelsListenV1MessagesListenV1ResultsMetadataDiarizeInfo:
type: object
properties:
model_uuid:
type: string
description: The diarizer model UUID
arch:
type: string
description: The diarizer arch, such as `v1` or `v2`
required:
- model_uuid
- arch
description: The diarizer that produced the speaker labels. Present only when a diarizer ran.
title: ChannelsListenV1MessagesListenV1ResultsMetadataDiarizeInfo
ChannelsListenV1MessagesListenV1ResultsMetadata:
type: object
properties:
request_id:
type: string
description: The request ID
model_info:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsMetadataModelInfo'
model_uuid:
type: string
description: The model UUID
diarize_info:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsMetadataDiarizeInfo'
description: The diarizer that produced the speaker labels. Present only when a diarizer ran.
required:
- request_id
- model_info
- model_uuid
title: ChannelsListenV1MessagesListenV1ResultsMetadata
ChannelsListenV1MessagesListenV1ResultsEntitiesItems:
type: object
properties:
label:
type: string
description: The type/category of the entity (e.g., NAME, PHONE_NUMBER, EMAIL_ADDRESS, ORGANIZATION, CARDINAL)
value:
type: string
description: The formatted text representation of the entity
raw_value:
type: string
description: The original spoken text of the entity (present when formatting is enabled)
confidence:
type: number
format: double
description: The confidence score of the entity detection
start_word:
type: integer
description: The index of the first word of the entity in the transcript (inclusive)
end_word:
type: integer
description: The index of the last word of the entity in the transcript (exclusive)
required:
- label
- value
- raw_value
- confidence
- start_word
- end_word
title: ChannelsListenV1MessagesListenV1ResultsEntitiesItems
ListenV1_ListenV1Results:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsType'
description: Message type identifier
channel_index:
type: array
items:
type: integer
description: The index of the channel
duration:
type: number
format: double
description: The duration of the transcription
start:
type: number
format: double
description: The start time of the transcription
is_final:
type: boolean
description: Whether the transcription is final
speech_final:
type: boolean
description: Whether the transcription is speech final
channel:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsChannel'
metadata:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsMetadata'
from_finalize:
type: boolean
description: Whether the transcription is from a finalize message
entities:
type: array
items:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1ResultsEntitiesItems'
description: >-
Extracted entities from the audio when detect_entities is enabled. Only present in is_final messages.
Returns an empty array if no entities are detected
required:
- type
- channel_index
- duration
- start
- channel
- metadata
title: ListenV1_ListenV1Results
ChannelsListenV1MessagesListenV1MetadataType:
type: string
enum:
- Metadata
description: Message type identifier
title: ChannelsListenV1MessagesListenV1MetadataType
ListenV1_ListenV1Metadata:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1MetadataType'
description: Message type identifier
transaction_key:
type: string
description: The transaction key
deprecated: true
request_id:
type: string
format: uuid
description: The request ID
sha256:
type: string
description: The sha256
created:
type: string
description: The created
duration:
type: number
format: double
description: The duration
channels:
type: integer
description: The channels
required:
- type
- transaction_key
- request_id
- sha256
- created
- duration
- channels
title: ListenV1_ListenV1Metadata
ChannelsListenV1MessagesListenV1UtteranceEndType:
type: string
enum:
- UtteranceEnd
description: Message type identifier
title: ChannelsListenV1MessagesListenV1UtteranceEndType
ListenV1_ListenV1UtteranceEnd:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1UtteranceEndType'
description: Message type identifier
channel:
type: array
items:
type: integer
description: The channel
last_word_end:
type: number
format: double
description: The last word end
required:
- type
- channel
- last_word_end
title: ListenV1_ListenV1UtteranceEnd
ChannelsListenV1MessagesListenV1SpeechStartedType:
type: string
enum:
- SpeechStarted
description: Message type identifier
title: ChannelsListenV1MessagesListenV1SpeechStartedType
ListenV1_ListenV1SpeechStarted:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1SpeechStartedType'
description: Message type identifier
channel:
type: array
items:
type: integer
description: The channel
timestamp:
type: number
format: double
description: The timestamp
required:
- type
- channel
- timestamp
title: ListenV1_ListenV1SpeechStarted
ListenV1_ListenV1Media:
type: string
format: binary
title: ListenV1_ListenV1Media
ChannelsListenV1MessagesListenV1FinalizeType:
type: string
enum:
- Finalize
- CloseStream
- KeepAlive
description: Message type identifier
title: ChannelsListenV1MessagesListenV1FinalizeType
ListenV1_ListenV1Finalize:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1FinalizeType'
description: Message type identifier
required:
- type
title: ListenV1_ListenV1Finalize
ChannelsListenV1MessagesListenV1CloseStreamType:
type: string
enum:
- Finalize
- CloseStream
- KeepAlive
description: Message type identifier
title: ChannelsListenV1MessagesListenV1CloseStreamType
ListenV1_ListenV1CloseStream:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1CloseStreamType'
description: Message type identifier
required:
- type
title: ListenV1_ListenV1CloseStream
ChannelsListenV1MessagesListenV1KeepAliveType:
type: string
enum:
- Finalize
- CloseStream
- KeepAlive
description: Message type identifier
title: ChannelsListenV1MessagesListenV1KeepAliveType
ListenV1_ListenV1KeepAlive:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV1MessagesListenV1KeepAliveType'
description: Message type identifier
required:
- type
title: ListenV1_ListenV1KeepAlive
ListenV2Model:
type: string
enum:
- flux-general-en
- flux-general-multi
description: Defines the AI model used to process submitted audio.
title: ListenV2Model
ListenV2Encoding:
type: string
enum:
- linear16
- linear32
- mulaw
- alaw
- opus
- ogg-opus
description: >-
Encoding of the audio stream. Required if sending non-containerized/raw audio. If sending containerized audio,
this parameter should be omitted.
title: ListenV2Encoding
ListenV2SampleRate:
description: Any type
title: ListenV2SampleRate
ListenV2EagerEotThreshold:
description: Any type
title: ListenV2EagerEotThreshold
ListenV2EotThreshold:
description: Any type
title: ListenV2EotThreshold
ListenV2EotTimeoutMs:
description: Any type
title: ListenV2EotTimeoutMs
ListenV2Keyterm:
oneOf:
- type: string
- type: array
items:
type: string
description: |
Keyterm prompting improves recognition of specialized terminology.
`keyterm` accepts plain terms only. Unlike the legacy `keywords` feature,
it does not support weights or intensifiers. Appending one
(for example, `keyterm=term:0.15`) is not rejected—the weight is
silently ignored and the entire value is treated as a literal keyterm.
To boost multiple separate keyterms, repeat the `keyterm` parameter
(for example, `keyterm=term1&keyterm=term2`). To boost one multi-word
phrase as a single keyterm, join the words with `%20` or `+`
(for example, `keyterm=customer%20service`). Do not separate keyterms
with commas, semicolons, or line breaks.
title: ListenV2Keyterm
ListenV2LanguageHint:
oneOf:
- type: string
- type: array
items:
type: string
description: |
Language hints constrain and prioritize language detection for the
flux-general-multi model. Pass multiple language_hint query parameters
to specify multiple language codes. Empty values are rejected.
Only valid when model is flux-general-multi.
title: ListenV2LanguageHint
ListenV2ProfanityFilter:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: >-
Profanity Filter looks for recognized profanity and converts it to the nearest recognized non-profane word or
removes it from the transcript completely.
title: ListenV2ProfanityFilter
ListenV2Numerals:
type: string
enum:
- 'true'
- 'false'
default: 'false'
description: Numerals converts numbers from written format to numerical format
title: ListenV2Numerals
ListenV2Redact:
type: string
enum:
- numbers
- aggressive_numbers
description: >-
Redaction removes sensitive information from your transcripts. On Flux, only `numbers` and `aggressive_numbers`
are supported.
title: ListenV2Redact
ListenV2MipOptOut:
description: Any type
title: ListenV2MipOptOut
ListenV2Tag:
description: Any type
title: ListenV2Tag
ChannelsListenV2MessagesListenV2ConnectedType:
type: string
enum:
- Connected
description: Message type identifier
title: ChannelsListenV2MessagesListenV2ConnectedType
ListenV2_ListenV2Connected:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2ConnectedType'
description: Message type identifier
request_id:
type: string
format: uuid
description: The unique identifier of the request
sequence_id:
type: integer
minimum: 0
description: |
Starts at `0` and increments for each message the server sends
to the client. This includes messages of other types, like
`TurnInfo` messages.
required:
- type
- request_id
- sequence_id
title: ListenV2_ListenV2Connected
ChannelsListenV2MessagesListenV2TurnInfoEvent:
type: string
enum:
- Update
- StartOfTurn
- EagerEndOfTurn
- TurnResumed
- EndOfTurn
description: >
The type of event being reported.
- **Update** - Additional audio has been transcribed, but the turn state hasn't changed
- **StartOfTurn** - The user has begun speaking for the first time in the turn
- **EagerEndOfTurn** - The system has moderate confidence that the user has finished speaking for the turn. This
is an opportunity to begin preparing an agent reply
- **TurnResumed** - The system detected that speech had ended and therefore sent an **EagerEndOfTurn** event,
but speech is actually continuing for this turn
- **EndOfTurn** - The user has finished speaking for the turn
title: ChannelsListenV2MessagesListenV2TurnInfoEvent
ChannelsListenV2MessagesListenV2TurnInfoWordsItems:
type: object
properties:
word:
type: string
description: The individual punctuated, properly-cased word from the transcript
confidence:
type: string
title: float
description: Confidence that this word was transcribed correctly
start:
type: number
format: double
minimum: 0
description: The start time of the word
end:
type: number
format: double
minimum: 0
description: The end time of the word
required:
- word
- confidence
title: ChannelsListenV2MessagesListenV2TurnInfoWordsItems
ListenV2_ListenV2TurnInfo:
type: object
properties:
type:
type: string
enum:
- TurnInfo
request_id:
type: string
format: uuid
description: The unique identifier of the request
sequence_id:
type: integer
minimum: 0
description: >
Starts at `0` and increments for each message the server sends to the client. This includes messages of
other types, like `Connected` messages.
event:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2TurnInfoEvent'
description: >
The type of event being reported.
- **Update** - Additional audio has been transcribed, but the turn state hasn't changed
- **StartOfTurn** - The user has begun speaking for the first time in the turn
- **EagerEndOfTurn** - The system has moderate confidence that the user has finished speaking for the turn.
This is an opportunity to begin preparing an agent reply
- **TurnResumed** - The system detected that speech had ended and therefore sent an **EagerEndOfTurn**
event, but speech is actually continuing for this turn
- **EndOfTurn** - The user has finished speaking for the turn
turn_index:
type: integer
minimum: 0
description: The index of the current turn
audio_window_start:
type: string
title: float
description: Start time in seconds of the audio range that was transcribed
audio_window_end:
type: string
title: float
description: End time in seconds of the audio range that was transcribed
transcript:
type: string
description: Text that was said over the course of the current turn
words:
type: array
items:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2TurnInfoWordsItems'
description: The words in the `transcript`
end_of_turn_confidence:
type: string
title: float
description: Confidence that no more speech is coming in this turn
trigger:
type: string
description: >
The cause of the turn ending. Present on every `EndOfTurn` event and only there.
- **model** - the turn ended by Flux's native end-of-turn detection
- **manual** - the turn ended because a `ForceEndTurn` message was sent
- **timeout** - the turn ended because `eot_timeout_ms` elapsed
This is an open enum. New values may be added over time, so clients must tolerate values they do not
recognize.
languages:
type: array
items:
type: string
description: |
Detected languages sorted by descending frequency in the
transcript. Only present when the flux-general-multi model
detects languages in the audio.
languages_hinted:
type: array
items:
type: string
description: |
The language hints that were supplied for this turn. Only
present when language hints are configured.
required:
- type
- request_id
- sequence_id
- event
- turn_index
- audio_window_start
- audio_window_end
- transcript
- words
- end_of_turn_confidence
description: Describes the current turn and latest state of the turn
title: ListenV2_ListenV2TurnInfo
ChannelsListenV2MessagesListenV2ConfigureSuccessThresholds:
type: object
properties:
eager_eot_threshold:
$ref: '#/components/schemas/ListenV2EagerEotThreshold'
eot_threshold:
$ref: '#/components/schemas/ListenV2EotThreshold'
default: '0.7'
eot_timeout_ms:
$ref: '#/components/schemas/ListenV2EotTimeoutMs'
default: '5000'
description: |
Updates each parameter, if it is supplied. If a particular threshold parameter
is not supplied, the configuration continues using the currently configured value.
title: ChannelsListenV2MessagesListenV2ConfigureSuccessThresholds
ListenV2ConfigureNumerals:
type: boolean
default: false
description: >-
Numerals converts numbers from written format to numerical format. Applies to transcripts Flux STT sends after
it processes the update.
title: ListenV2ConfigureNumerals
ListenV2_ListenV2ConfigureSuccess:
type: object
properties:
type:
type: string
enum:
- ConfigureSuccess
description: Message type identifier
request_id:
type: string
format: uuid
description: The unique identifier of the request
thresholds:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2ConfigureSuccessThresholds'
description: |
Updates each parameter, if it is supplied. If a particular threshold parameter
is not supplied, the configuration continues using the currently configured value.
keyterms:
$ref: '#/components/schemas/ListenV2Keyterm'
language_hints:
type: array
items:
type: string
description: |
The currently active language hints. Only applicable to the flux-general-multi model.
numerals:
$ref: '#/components/schemas/ListenV2ConfigureNumerals'
default: false
description: Whether numeral formatting is enabled for transcripts Flux STT sends after it processes the update.
sequence_id:
type: integer
minimum: 0
description: |
Starts at `0` and increments for each message the server sends
to the client. This includes messages of other types, like
`TurnInfo` messages.
required:
- type
- request_id
- thresholds
- keyterms
- sequence_id
title: ListenV2_ListenV2ConfigureSuccess
ListenV2_ListenV2ConfigureFailure:
type: object
properties:
type:
type: string
enum:
- ConfigureFailure
description: Message type identifier
request_id:
type: string
format: uuid
description: The unique identifier of the request
sequence_id:
type: integer
minimum: 0
description: |
Starts at `0` and increments for each message the server sends
to the client. This includes messages of other types, like
`TurnInfo` messages.
code:
type: string
description: Failure code identifying the rejected configuration
description:
type: string
description: A human-readable description of the configuration failure
required:
- type
- request_id
- sequence_id
title: ListenV2_ListenV2ConfigureFailure
ListenV2_ListenV2Warning:
type: object
properties:
type:
type: string
enum:
- Warning
description: Message type identifier
request_id:
type: string
format: uuid
description: The unique identifier of the request
sequence_id:
type: integer
minimum: 0
description: |
Starts at `0` and increments for each message the server sends
to the client. This includes messages of other types, like
`TurnInfo` messages.
code:
type: string
description: Warning code identifying the condition, in `SCREAMING_SNAKE_CASE`
description:
type: string
description: A human-readable description of the warning
required:
- type
- request_id
- sequence_id
- code
- description
title: ListenV2_ListenV2Warning
ChannelsListenV2MessagesListenV2FatalErrorType:
type: string
enum:
- Error
description: Message type identifier
title: ChannelsListenV2MessagesListenV2FatalErrorType
ListenV2_ListenV2FatalError:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2FatalErrorType'
description: Message type identifier
sequence_id:
type: integer
minimum: 0
description: |
Starts at `0` and increments for each message the server sends
to the client. This includes messages of other types, like
`Connected` messages.
code:
type: string
description: A string code describing the error, e.g. `INTERNAL_SERVER_ERROR`
description:
type: string
description: Prose description of the error
required:
- type
- sequence_id
- code
- description
title: ListenV2_ListenV2FatalError
ListenV2_ListenV2Media:
type: string
format: binary
title: ListenV2_ListenV2Media
ChannelsListenV2MessagesListenV2CloseStreamType:
type: string
enum:
- CloseStream
description: Message type identifier
title: ChannelsListenV2MessagesListenV2CloseStreamType
ListenV2_ListenV2CloseStream:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2CloseStreamType'
description: Message type identifier
required:
- type
title: ListenV2_ListenV2CloseStream
ListenV2_ListenV2ForceEndTurn:
type: object
properties:
type:
type: string
enum:
- ForceEndTurn
description: Message type identifier
required:
- type
title: ListenV2_ListenV2ForceEndTurn
ChannelsListenV2MessagesListenV2ConfigureThresholds:
type: object
properties:
eager_eot_threshold:
$ref: '#/components/schemas/ListenV2EagerEotThreshold'
eot_threshold:
$ref: '#/components/schemas/ListenV2EotThreshold'
default: '0.7'
eot_timeout_ms:
$ref: '#/components/schemas/ListenV2EotTimeoutMs'
default: '5000'
description: |
Updates each parameter, if it is supplied. If a particular threshold parameter
is not supplied, the configuration continues using the currently configured value.
title: ChannelsListenV2MessagesListenV2ConfigureThresholds
ListenV2_ListenV2Configure:
type: object
properties:
type:
type: string
enum:
- Configure
description: Message type identifier
thresholds:
$ref: '#/components/schemas/ChannelsListenV2MessagesListenV2ConfigureThresholds'
description: |
Updates each parameter, if it is supplied. If a particular threshold parameter
is not supplied, the configuration continues using the currently configured value.
keyterms:
$ref: '#/components/schemas/ListenV2Keyterm'
language_hints:
type: array
items:
type: string
description: |
Language hints to constrain and prioritize language detection.
Only valid when the model is flux-general-multi. If this field is not supplied,
the session will continue to use the currently configured value.
numerals:
$ref: '#/components/schemas/ListenV2ConfigureNumerals'
default: false
required:
- type
title: ListenV2_ListenV2Configure
SpeakV1Encoding:
type: string
enum:
- linear16
- mulaw
- alaw
default: linear16
description: >-
Encoding allows you to specify the expected encoding of your audio output for streaming TTS. Only
streaming-compatible encodings are supported.
title: SpeakV1Encoding
SpeakV1MipOptOut:
description: Any type
title: SpeakV1MipOptOut
SpeakV1Model:
type: string
enum:
- aura-angus-en
- aura-arcas-en
- aura-asteria-en
- aura-athena-en
- aura-helios-en
- aura-hera-en
- aura-luna-en
- aura-orion-en
- aura-orpheus-en
- aura-perseus-en
- aura-stella-en
- aura-zeus-en
- aura-2-amalthea-en
- aura-2-andromeda-en
- aura-2-apollo-en
- aura-2-arcas-en
- aura-2-aries-en
- aura-2-asteria-en
- aura-2-athena-en
- aura-2-atlas-en
- aura-2-aurora-en
- aura-2-callista-en
- aura-2-cora-en
- aura-2-cordelia-en
- aura-2-delia-en
- aura-2-draco-en
- aura-2-electra-en
- aura-2-harmonia-en
- aura-2-helena-en
- aura-2-hera-en
- aura-2-hermes-en
- aura-2-hyperion-en
- aura-2-iris-en
- aura-2-janus-en
- aura-2-juno-en
- aura-2-jupiter-en
- aura-2-luna-en
- aura-2-mars-en
- aura-2-minerva-en
- aura-2-neptune-en
- aura-2-odysseus-en
- aura-2-ophelia-en
- aura-2-orion-en
- aura-2-orpheus-en
- aura-2-pandora-en
- aura-2-phoebe-en
- aura-2-pluto-en
- aura-2-saturn-en
- aura-2-selene-en
- aura-2-thalia-en
- aura-2-theia-en
- aura-2-vesta-en
- aura-2-zeus-en
- aura-2-agustina-es
- aura-2-alvaro-es
- aura-2-antonia-es
- aura-2-aquila-es
- aura-2-carina-es
- aura-2-celeste-es
- aura-2-diana-es
- aura-2-estrella-es
- aura-2-gloria-es
- aura-2-javier-es
- aura-2-luciano-es
- aura-2-nestor-es
- aura-2-olivia-es
- aura-2-selena-es
- aura-2-silvia-es
- aura-2-sirio-es
- aura-2-valerio-es
- aura-2-aurelia-de
- aura-2-elara-de
- aura-2-fabian-de
- aura-2-julius-de
- aura-2-kara-de
- aura-2-lara-de
- aura-2-viktoria-de
- aura-2-beatrix-nl
- aura-2-cornelia-nl
- aura-2-daphne-nl
- aura-2-hestia-nl
- aura-2-lars-nl
- aura-2-leda-nl
- aura-2-rhea-nl
- aura-2-roman-nl
- aura-2-sander-nl
- aura-2-agathe-fr
- aura-2-hector-fr
- aura-2-cesare-it
- aura-2-cinzia-it
- aura-2-demetra-it
- aura-2-dionisio-it
- aura-2-elio-it
- aura-2-flavio-it
- aura-2-livia-it
- aura-2-maia-it
- aura-2-melia-it
- aura-2-ama-ja
- aura-2-ebisu-ja
- aura-2-fujin-ja
- aura-2-izanami-ja
- aura-2-uzume-ja
default: aura-asteria-en
description: AI model used to process submitted text
title: SpeakV1Model
SpeakV1SampleRate:
type: string
enum:
- '8000'
- '16000'
- '24000'
- '32000'
- '48000'
default: '24000'
description: >-
Sample Rate specifies the sample rate for the output audio. Based on encoding 8000 or 24000 are possible
defaults. For some encodings sample rate is not configurable.
title: SpeakV1SampleRate
SpeakV1Speed:
type: number
format: double
minimum: 0.7
maximum: 1.5
default: 1
description: >-
Speaking rate multiplier that adjusts the pace of generated speech while preserving natural prosody and voice
quality. Not yet supported in all languages.
title: SpeakV1Speed
SpeakV1_SpeakV1Audio:
type: string
format: binary
title: SpeakV1_SpeakV1Audio
ChannelsSpeakV1MessagesSpeakV1MetadataType:
type: string
enum:
- Metadata
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1MetadataType
SpeakV1_SpeakV1Metadata:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1MetadataType'
description: Message type identifier
request_id:
type: string
format: uuid
description: Unique identifier for the request
model_name:
type: string
description: Name of the model being used
model_version:
type: string
description: Version of the primary model being used
model_uuid:
type: string
format: uuid
description: Unique identifier for the primary model used
additional_model_uuids:
type: array
items:
type: string
format: uuid
description: List of unique identifiers for any additional models used to serve the request
required:
- type
- request_id
- model_name
- model_version
- model_uuid
title: SpeakV1_SpeakV1Metadata
ChannelsSpeakV1MessagesSpeakV1FlushedType:
type: string
enum:
- Flushed
- Cleared
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1FlushedType
SpeakV1_SpeakV1Flushed:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1FlushedType'
description: Message type identifier
sequence_id:
type: integer
description: The sequence ID of the response
required:
- type
- sequence_id
title: SpeakV1_SpeakV1Flushed
ChannelsSpeakV1MessagesSpeakV1ClearedType:
type: string
enum:
- Flushed
- Cleared
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1ClearedType
SpeakV1_SpeakV1Cleared:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1ClearedType'
description: Message type identifier
sequence_id:
type: integer
description: The sequence ID of the response
required:
- type
- sequence_id
title: SpeakV1_SpeakV1Cleared
ChannelsSpeakV1MessagesSpeakV1WarningType:
type: string
enum:
- Warning
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1WarningType
SpeakV1_SpeakV1Warning:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1WarningType'
description: Message type identifier
description:
type: string
description: A description of what went wrong
code:
type: string
description: Error code identifying the type of error
required:
- type
- description
- code
title: SpeakV1_SpeakV1Warning
ChannelsSpeakV1MessagesSpeakV1TextType:
type: string
enum:
- Speak
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1TextType
SpeakV1_SpeakV1Text:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1TextType'
description: Message type identifier
text:
type: string
description: The input text to be converted to speech
required:
- type
- text
title: SpeakV1_SpeakV1Text
ChannelsSpeakV1MessagesSpeakV1FlushType:
type: string
enum:
- Flush
- Clear
- Close
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1FlushType
SpeakV1_SpeakV1Flush:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1FlushType'
description: Message type identifier
required:
- type
title: SpeakV1_SpeakV1Flush
ChannelsSpeakV1MessagesSpeakV1ClearType:
type: string
enum:
- Flush
- Clear
- Close
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1ClearType
SpeakV1_SpeakV1Clear:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1ClearType'
description: Message type identifier
required:
- type
title: SpeakV1_SpeakV1Clear
ChannelsSpeakV1MessagesSpeakV1CloseType:
type: string
enum:
- Flush
- Clear
- Close
description: Message type identifier
title: ChannelsSpeakV1MessagesSpeakV1CloseType
SpeakV1_SpeakV1Close:
type: object
properties:
type:
$ref: '#/components/schemas/ChannelsSpeakV1MessagesSpeakV1CloseType'
description: Message type identifier
required:
- type
title: SpeakV1_SpeakV1Close
SpeakV2Model:
type: string
description: >-
The Flux TTS model used to synthesize speech. Required on every connection. Model strings follow the format
`flux-{voice}-{language}` (e.g. `flux-alexis-en`). An Aura model string is rejected on `/v2/speak`; use
`/v1/speak` for Aura voices.
title: SpeakV2Model
SpeakV2Encoding:
type: string
enum:
- linear16
- mulaw
- alaw
default: linear16
description: >-
Encoding of the raw output audio. The streaming WebSocket emits raw (non-containerized) audio, so only
streaming-compatible encodings are supported. Compressed and containerized encodings (`mp3`, `opus`, `flac`,
`aac`) are available on the batch REST transport only.
title: SpeakV2Encoding
SpeakV2SampleRate:
type: string
enum:
- '8000'
- '16000'
- '24000'
- '32000'
- '44100'
- '48000'
description: >-
Output sample rate in Hz. With `linear16`, valid values are `8000`, `16000`, `24000`, `32000`, `44100`, and
`48000`. With `mulaw` or `alaw`, valid values are `8000` and `16000`. Defaults to the model's native sample
rate.
title: SpeakV2SampleRate
SpeakV2Speed:
type: number
format: double
minimum: 0.5
maximum: 1.5
multipleOf: 0.05
default: 1
description: >-
Speech-rate multiplier. `1.0` is the model's nominal rate; lower is slower. Accepted values run `0.5` to `1.5`
in `0.05` increments. A value outside that range is rejected with `SPEED_OUT_OF_RANGE`; a value inside it but
off the `0.05` increment with `SPEED_INCREMENT_INVALID`. Models and languages without runtime speed control
reject any value with `SPEED_NOT_SUPPORTED`.
title: SpeakV2Speed
SpeakV2Expressivity:
type: string
enum:
- '-2'
- '-1'
- '0'
- '1'
- '2'
description: >-
Expressive range of the generated speech, on a calm-to-animated axis. Accepted values: `-2`, `-1`, `0`, `1`,
`2`. `0` (the default) is the voice's tuned delivery and the production-validated setting, with `-2` the calm
end of the range and `2` the animated end. Supported on all Flux voices. Fixed for the connection — not settable
via `Configure`. Beta: behavior may change in future model versions, and non-default values increase the risk of
hallucinations and pronunciation errors; audition before shipping. An invalid value fails the connection with a
`400` — `EXPRESSIVITY_OUT_OF_RANGE` for a value outside the range, `EXPRESSIVITY_INCREMENT_INVALID` for a
fractional value. See [Expressivity](/guides/tts-voice-controls-tts-expressivity).
title: SpeakV2Expressivity
SpeakV2MipOptOut:
description: Any type
title: SpeakV2MipOptOut
SpeakV2Tag:
description: Any type
title: SpeakV2Tag
SpeakV2_SpeakV2Audio:
type: string
format: binary
title: SpeakV2_SpeakV2Audio
SpeakV2_SpeakV2Connected:
type: object
properties:
type:
type: string
enum:
- Connected
description: Message type identifier
request_id:
type: string
format: uuid
description: The unique identifier of the `/v2/speak` request
model_name:
type: string
description: Resolved model name
model_version:
type: string
description: Resolved model version
model_uuids:
type: array
items:
type: string
format: uuid
description: Resolved model UUIDs. A list, because a resolved model may be backed by more than one underlying model.
required:
- type
- request_id
- model_name
- model_version
- model_uuids
title: SpeakV2_SpeakV2Connected
SpeakV2_SpeakV2SpeechStarted:
type: object
properties:
type:
type: string
enum:
- SpeechStarted
description: Message type identifier
speech_id:
type: string
description: Server-minted identifier for this turn, of the form `dg_sp_<12 hex digits>`. Informational.
required:
- type
- speech_id
title: SpeakV2_SpeakV2SpeechStarted
ChannelsSpeakV2MessagesSpeakV2SpeechMetadataControlsApplied:
type: object
properties:
pronunciations_applied:
type: integer
description: >-
Pronunciation overrides successfully applied. Mirrors the Aura-2 `dg-pronunciations-applied` REST header.
Currently always `0`.
breaks_applied:
type: integer
description: >-
Pause (break) controls successfully applied. Mirrors the Aura-2 `dg-breaks-applied` REST header. Currently
always `0`.
pronunciation_warnings:
type: integer
description: >-
Pronunciation entries that triggered a warning (invalid IPA, word too long). Mirrors the Aura-2
`dg-pronunciation-warnings` REST header. Currently always `0`.
required:
- pronunciations_applied
- breaks_applied
- pronunciation_warnings
description: >-
Counts of the inline controls the server acted on during the turn. Inline pause and pronunciation controls are
not applied at launch — support is coming soon — so every count is currently `0`.
title: ChannelsSpeakV2MessagesSpeakV2SpeechMetadataControlsApplied
SpeakV2_SpeakV2SpeechMetadata:
type: object
properties:
type:
type: string
enum:
- SpeechMetadata
description: Message type identifier
speech_id:
type: string
description: Server-assigned turn identifier
audio_duration_ms:
type: integer
description: Total audio duration produced for this turn, in milliseconds
input_character_count:
type: integer
description: Raw input character count for this turn, before text normalization
billable_character_count:
type: integer
description: >-
Billable character count for this turn — the input character count with stripped control characters removed.
Always less than or equal to `input_character_count`.
controls_applied:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2SpeechMetadataControlsApplied'
description: >-
Counts of the inline controls the server acted on during the turn. Inline pause and pronunciation controls
are not applied at launch — support is coming soon — so every count is currently `0`.
required:
- type
- speech_id
- audio_duration_ms
- input_character_count
- billable_character_count
- controls_applied
title: SpeakV2_SpeakV2SpeechMetadata
ChannelsSpeakV2MessagesSpeakV2SpeechInterruptedMetadataControlsApplied:
type: object
properties:
pronunciations_applied:
type: integer
description: >-
Pronunciation overrides successfully applied. Mirrors the Aura-2 `dg-pronunciations-applied` REST header.
Currently always `0`.
breaks_applied:
type: integer
description: >-
Pause (break) controls successfully applied. Mirrors the Aura-2 `dg-breaks-applied` REST header. Currently
always `0`.
pronunciation_warnings:
type: integer
description: >-
Pronunciation entries that triggered a warning (invalid IPA, word too long). Mirrors the Aura-2
`dg-pronunciation-warnings` REST header. Currently always `0`.
required:
- pronunciations_applied
- breaks_applied
- pronunciation_warnings
description: >-
Counts of the inline controls the server acted on during the turn. Inline pause and pronunciation controls are
not applied at launch — support is coming soon — so every count is currently `0`.
title: ChannelsSpeakV2MessagesSpeakV2SpeechInterruptedMetadataControlsApplied
ChannelsSpeakV2MessagesSpeakV2SpeechInterruptedMetadata:
type: object
properties:
speech_id:
type: string
description: Server-assigned turn identifier
audio_duration_ms:
type: integer
description: Audio duration produced for this turn, in milliseconds
input_character_count:
type: integer
description: Raw input character count for this turn, before text normalization
billable_character_count:
type: integer
description: >-
Billable character count for this turn — the input character count with stripped control characters removed.
Always less than or equal to `input_character_count`.
controls_applied:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2SpeechInterruptedMetadataControlsApplied'
description: >-
Counts of the inline controls the server acted on during the turn. Inline pause and pronunciation controls
are not applied at launch — support is coming soon — so every count is currently `0`.
required:
- speech_id
- audio_duration_ms
- input_character_count
- billable_character_count
- controls_applied
description: Billing and timing for a single turn.
title: ChannelsSpeakV2MessagesSpeakV2SpeechInterruptedMetadata
SpeakV2_SpeakV2SpeechInterrupted:
type: object
properties:
type:
type: string
enum:
- SpeechInterrupted
description: Message type identifier
audio_played_ms:
type: integer
description: >-
How much audio the client had played when the interrupt landed, in milliseconds from the start of the
session. Echoes the `Interrupt`'s `playback_offset` when one was supplied. Otherwise it is the server's own
total, representing the audio that has been generated so far. A client that sends its first `Interrupt`
without an offset can use this value as the baseline the next one must advance past.
text_spoken:
type: string
description: The portion of the turn's text the user heard. Omitted when the `Interrupt` carried no `playback_offset`.
text_remaining:
type: string
description: >-
The portion of the turn's text the user did not hear. Omitted when the `Interrupt` carried no
`playback_offset`.
metadata:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2SpeechInterruptedMetadata'
description: Billing and timing for a single turn.
required:
- type
- audio_played_ms
- metadata
title: SpeakV2_SpeakV2SpeechInterrupted
SpeakV2_SpeakV2Flushed:
type: object
properties:
type:
type: string
enum:
- Flushed
description: Message type identifier
speech_id:
type: string
description: Server-assigned turn identifier
required:
- type
- speech_id
title: SpeakV2_SpeakV2Flushed
SpeakV2_SpeakV2SessionMetadata:
type: object
properties:
type:
type: string
enum:
- SessionMetadata
description: Message type identifier
total_audio_duration_ms:
type: integer
description: >-
Cumulative audio duration produced across the session, in milliseconds. An `Interrupt` rebases this onto the
audio the client actually played.
total_input_character_count:
type: integer
description: Cumulative raw input character count across the session
total_billable_character_count:
type: integer
description: Cumulative billable character count across the session
required:
- type
- total_audio_duration_ms
- total_input_character_count
- total_billable_character_count
title: SpeakV2_SpeakV2SessionMetadata
SpeakV2SpeedValue:
type: number
format: double
minimum: 0.5
maximum: 1.5
multipleOf: 0.05
default: 1
description: >-
Speech-rate multiplier. `1.0` is the model's nominal rate; lower is slower. Accepted values run `0.5` to `1.5`
in `0.05` increments. A value outside that range is rejected with `SPEED_OUT_OF_RANGE`; a value inside it but
off the `0.05` increment with `SPEED_INCREMENT_INVALID`. Models and languages without runtime speed control
reject any value with `SPEED_NOT_SUPPORTED`.
title: SpeakV2SpeedValue
ChannelsSpeakV2MessagesSpeakV2ConfigureSuccessApplied:
type: object
properties:
speed:
$ref: '#/components/schemas/SpeakV2SpeedValue'
default: 1
description: Synthesis configuration. A field is present only when it has been set on this session.
title: ChannelsSpeakV2MessagesSpeakV2ConfigureSuccessApplied
SpeakV2_SpeakV2ConfigureSuccess:
type: object
properties:
type:
type: string
enum:
- ConfigureSuccess
description: Message type identifier
applied:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2ConfigureSuccessApplied'
description: Synthesis configuration. A field is present only when it has been set on this session.
required:
- type
- applied
title: SpeakV2_SpeakV2ConfigureSuccess
ChannelsSpeakV2MessagesSpeakV2ConfigureFailureCode:
type: string
enum:
- SPEED_OUT_OF_RANGE
- SPEED_INCREMENT_INVALID
- SPEED_NOT_SUPPORTED
- INTERNAL_ERROR
description: >-
Failure code, in `SCREAMING_SNAKE_CASE`. `SPEED_OUT_OF_RANGE`: outside the range the model publishes.
`SPEED_INCREMENT_INVALID`: inside the published range but off the `0.05` increment. `SPEED_NOT_SUPPORTED`: this
model or language has no runtime speed control at all. `INTERNAL_ERROR`: the configuration was acceptable but
the server could not apply it — unlike the others, a server-side failure rather than a statement about the
request.
title: ChannelsSpeakV2MessagesSpeakV2ConfigureFailureCode
ChannelsSpeakV2MessagesSpeakV2ConfigureFailureField:
type: string
enum:
- speed
description: The configuration field the failure is about. Absent when the failure is not tied to one field.
title: ChannelsSpeakV2MessagesSpeakV2ConfigureFailureField
SpeakV2_SpeakV2ConfigureFailure:
type: object
properties:
type:
type: string
enum:
- ConfigureFailure
description: Message type identifier
code:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2ConfigureFailureCode'
description: >-
Failure code, in `SCREAMING_SNAKE_CASE`. `SPEED_OUT_OF_RANGE`: outside the range the model publishes.
`SPEED_INCREMENT_INVALID`: inside the published range but off the `0.05` increment. `SPEED_NOT_SUPPORTED`:
this model or language has no runtime speed control at all. `INTERNAL_ERROR`: the configuration was
acceptable but the server could not apply it — unlike the others, a server-side failure rather than a
statement about the request.
field:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2ConfigureFailureField'
description: The configuration field the failure is about. Absent when the failure is not tied to one field.
value:
type: number
format: double
description: >-
The rejected value for `field`. Absent when there is no offending value to echo — `SPEED_NOT_SUPPORTED`
names the field but carries no value, because the rejection is a property of the model.
description:
type: string
description: A human-readable description of the failure
required:
- type
- code
- description
title: SpeakV2_SpeakV2ConfigureFailure
SpeakV2_SpeakV2Warning:
type: object
properties:
type:
type: string
enum:
- Warning
description: Message type identifier
code:
type: string
description: >-
Warning code identifying the condition, in `SCREAMING_SNAKE_CASE`.
Turn-scoped codes: `NO_ACTIVE_SPEECH` (a speech-scoped message arrived with no active turn),
`NO_SYNTHESIZABLE_TEXT` (the turn's text was entirely whitespace or punctuation, so it produced no audio and
is completed with a zero-duration `SpeechMetadata`), and `SYNTHESIS_RETRYING` (a synthesis request failed
and is being retried).
Inline-control codes are reserved and not currently emitted, because inline pause and pronunciation controls
are not yet applied: `BREAKS_LIMIT_EXCEEDED` (too many pause controls, or two pauses with no intervening
text), `BREAK_TOKENS_OUT_OF_RANGE` (pause durations outside the range the model supports),
`BREAK_TOKENS_WITH_INVALID_INCREMENTS` (pause durations off the model's supported increment),
`PRONUNCIATION_WARNINGS` (a pronunciation override contained invalid IPA), `PRONUNCIATION_TOO_LONG` (an IPA
string exceeded the length limit), `PRONUNCIATIONS_LIMIT_EXCEEDED` (too many pronunciation controls in one
turn).
Interrupt-scoped codes, each meaning the `Interrupt` was ignored: `NO_AUDIO_GENERATED` (the session has
produced no audio yet, so there is nothing to interrupt), `INTERRUPT_IN_PROGRESS` (an earlier `Interrupt` is
still being processed — at most one is handled at a time), `INVALID_INTERRUPT_OFFSET` (the `playback_offset`
did not advance past the position a prior interrupt established).
description:
type: string
description: A human-readable description of the warning
required:
- type
- code
- description
title: SpeakV2_SpeakV2Warning
ChannelsSpeakV2MessagesSpeakV2ErrorCode:
type: string
enum:
- MESSAGE-0000
- DATA-0000
- DATA-0002
- BIG-0000
- NET-0000
- NET-0001
- NET-0002
- NET-0003
- NET-0004
description: A code identifying the error, e.g. `MESSAGE-0000` or `NET-0000`.
title: ChannelsSpeakV2MessagesSpeakV2ErrorCode
SpeakV2_SpeakV2Error:
type: object
properties:
type:
type: string
enum:
- Error
description: Message type identifier
code:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2ErrorCode'
description: A code identifying the error, e.g. `MESSAGE-0000` or `NET-0000`.
description:
type: string
description: Prose description of the error
required:
- type
- code
- description
title: SpeakV2_SpeakV2Error
SpeakV2_SpeakV2Speak:
type: object
properties:
type:
type: string
enum:
- Speak
description: Message type identifier
text:
type: string
description: >-
The input text to synthesize. Inline pause and pronunciation controls are not yet applied; they are stripped
from the text before synthesis.
required:
- type
- text
title: SpeakV2_SpeakV2Speak
SpeakV2_SpeakV2Flush:
type: object
properties:
type:
type: string
enum:
- Flush
description: Message type identifier
required:
- type
title: SpeakV2_SpeakV2Flush
ChannelsSpeakV2MessagesSpeakV2InterruptPlaybackOffset:
type: object
properties:
type:
type: string
enum:
- time_ms
description: Offset unit. `time_ms` is the only supported form.
value:
type: integer
description: Milliseconds of session audio the client played before barging in.
required:
- type
- value
description: >-
How much audio the client had played when the user barged in. Optional: without it the server cannot split the
turn's text, so `SpeechInterrupted` omits `text_spoken` and `text_remaining`.
The offset is cumulative from the start of the *session*, not from the start of the current turn. Each
`Interrupt` must advance past the position the previous one established.
title: ChannelsSpeakV2MessagesSpeakV2InterruptPlaybackOffset
SpeakV2_SpeakV2Interrupt:
type: object
properties:
type:
type: string
enum:
- Interrupt
description: Message type identifier
playback_offset:
$ref: '#/components/schemas/ChannelsSpeakV2MessagesSpeakV2InterruptPlaybackOffset'
description: >-
How much audio the client had played when the user barged in. Optional: without it the server cannot split
the turn's text, so `SpeechInterrupted` omits `text_spoken` and `text_remaining`.
The offset is cumulative from the start of the *session*, not from the start of the current turn. Each
`Interrupt` must advance past the position the previous one established.
required:
- type
title: SpeakV2_SpeakV2Interrupt
SpeakV2_SpeakV2Configure:
type: object
properties:
type:
type: string
enum:
- Configure
description: Message type identifier
speed:
$ref: '#/components/schemas/SpeakV2SpeedValue'
default: 1
required:
- type
title: SpeakV2_SpeakV2Configure
SpeakV2_SpeakV2Close:
type: object
properties:
type:
type: string
enum:
- Close
description: Message type identifier
required:
- type
title: SpeakV2_SpeakV2Close