From 07465698647aa7751bd30562896eba2a6436ce8e Mon Sep 17 00:00:00 2001 From: Him188 Date: Mon, 28 Sep 2026 17:47:03 +0900 Subject: [PATCH 1/3] agents: document the asr config section and strict language Speech recognition settings are configured through the asr section (asr.model, asr.multilingual), and a new Strict language section covers asr.strict_language with its per-model behavior and misdetection risk. Co-Authored-By: Claude Opus 5.5 --- agents/build/configuration.mdx | 9 +++++++- agents/build/voice-language.mdx | 41 +++++++++++++++++++++++++++------ 2 files changed, 42 insertions(+), 8 deletions(-) diff --git a/agents/build/configuration.mdx b/agents/build/configuration.mdx index 9f4758c..ba2f1cd 100644 --- a/agents/build/configuration.mdx +++ b/agents/build/configuration.mdx @@ -34,7 +34,7 @@ Choose how the agent opens each conversation: ## Voice -The Voice panel selects the voice your agent speaks with (`voice_id`) and its speaking language (`speaking_language`, one of the [52 supported languages](/agents/build/voice-language#speaking-language)). It also holds the speech recognition model (`asr_model`), the multilingual recognition switch (`multilingual_asr`), and expressive delivery (`expressive`). See [Voice & language](/agents/build/voice-language) for all of them. +The Voice panel selects the voice your agent speaks with (`voice_id`) and its speaking language (`speaking_language`, one of the [52 supported languages](/agents/build/voice-language#speaking-language)). It also holds expressive delivery (`expressive`) and the speech recognition settings, which the API keeps in their own `asr` section: the recognition model (`asr.model`) and the multilingual recognition switch (`asr.multilingual`). See [Voice & language](/agents/build/voice-language) for all of them. ## Conversation settings @@ -106,6 +106,12 @@ curl --request GET "https://api.fish.audio/v1/agent/agents/$AGENT_ID/config" \ "voice_id": "802e3bc2b27e49c2995d23ef70e6ac89", "speaking_language": "en" }, + "asr": { + "model": "deepgram:nova-3", + "multilingual": false, + "strict_language": false, + "keyterms": [] + }, "conversation": { "max_duration_seconds": 1800, "response_wait_ms": 550, @@ -144,6 +150,7 @@ The response includes the draft's new `config_hash`. Values outside the document |---|---|---| | `prompt` | System prompt and first-message settings | This page | | `voice` | Voice profile, speaking language, and expressive mode | [Voice & language](/agents/build/voice-language) | +| `asr` | Speech recognition model, multilingual recognition, strict language, and recognition keyterms | [Voice & language](/agents/build/voice-language#speech-recognition-model) | | `conversation` | Call duration, turn-taking, interruption behavior, timezone, and storage | This page; storage in [Conversation history](/agents/monitor/conversation-history#what-gets-stored) | | `tools` | Attached webhook tools and system tool switches | [Tools](/agents/build/tools) | | `knowledge_base` | Attached knowledge sources | [Knowledge base](/agents/build/knowledge-base) | diff --git a/agents/build/voice-language.mdx b/agents/build/voice-language.mdx index 3f8d72f..42cedbe 100644 --- a/agents/build/voice-language.mdx +++ b/agents/build/voice-language.mdx @@ -106,7 +106,7 @@ Both settings can also be replaced for a single session: send `overrides.voice_i ## Speech recognition model -**Speech recognition** picks the model that transcribes what callers say (`voice.asr_model` on the wire). Latency is the typical wait from the caller's last word to the transcript. The agent starts its reply after that. +**Speech recognition** picks the model that transcribes what callers say (`asr.model` on the wire). Latency is the typical wait from the caller's last word to the transcript. The agent starts its reply after that. | Value | Model | Latency | Notes | |---|---|---|---| @@ -122,8 +122,8 @@ curl --request PATCH "https://api.fish.audio/v1/agent/agents/$AGENT_ID/config" \ --header "Authorization: Bearer $FISH_API_KEY" \ --header "Content-Type: application/json" \ --data '{ - "voice": { - "asr_model": "elevenlabs:scribe_v2_realtime" + "asr": { + "model": "elevenlabs:scribe_v2_realtime" } }' ``` @@ -135,7 +135,7 @@ Speech recognition applies to spoken sessions (voice calls and phone). As with e **Multilingual recognition** lets the agent understand callers who switch to another language partway through a conversation. With it off, speech recognition listens only for the speaking language. That is more accurate when every caller speaks the same language, and it keeps short or accented phrases from being transcribed as another language. -It is off by default. Toggle it with the **Multilingual recognition** switch in the Builder's voice section, or via the API (`voice.multilingual_asr`, default `false`): +It is off by default. Toggle it with the **Multilingual recognition** switch in the Builder's voice section, or via the API (`asr.multilingual`, default `false`): ```bash API (curl) @@ -143,17 +143,44 @@ curl --request PATCH "https://api.fish.audio/v1/agent/agents/$AGENT_ID/config" \ --header "Authorization: Bearer $FISH_API_KEY" \ --header "Content-Type: application/json" \ --data '{ - "voice": { - "multilingual_asr": true + "asr": { + "multilingual": true + } + }' +``` + + + + Which languages a caller can switch between depends on the [speech recognition model](#speech-recognition-model). With Deepgram Nova-3, a speaking language outside its switching set stays locked even with the setting on. With ElevenLabs Scribe v2, turning the setting off steers recognition toward the speaking language rather than locking it. To lock it, use [strict language](#strict-language). + + +## Strict language + +**Strict language** keeps the agent from understanding any language other than its speaking language. When speech recognition detects that the caller spoke another language, the agent receives `[unintelligible speech]` instead of the transcript and answers as if it did not catch what was said. It applies whether multilingual recognition is on or off. + +It is off by default. Turn it on via the API (`asr.strict_language`, default `false`): + + +```bash API (curl) +curl --request PATCH "https://api.fish.audio/v1/agent/agents/$AGENT_ID/config" \ + --header "Authorization: Bearer $FISH_API_KEY" \ + --header "Content-Type: application/json" \ + --data '{ + "asr": { + "strict_language": true } }' ``` - Which languages a caller can switch between depends on the [speech recognition model](#speech-recognition-model). With Deepgram Nova-3, a speaking language outside its switching set stays locked even with the setting on. With ElevenLabs Scribe v2, turning the setting off steers recognition toward the speaking language rather than locking it. + With the ElevenLabs Scribe v2 models, speech in another language reaches the agent as `[unintelligible speech]`. With Deepgram Nova-3, recognition runs a model for the speaking language only, so speech in another language is usually not transcribed at all and the agent keeps listening. + + Language is detected per utterance, so a short or heavily accented phrase in the speaking language can occasionally be detected as another language and replaced with `[unintelligible speech]` too. Turn strict language on only when callers are expected to speak the speaking language. + + ## Speaking speed **Speaking speed** sets how fast the agent talks, as a multiplier from `0.5` (half speed) to `2.0` (double speed). The default is `1.0`. Many English-language agents sound more natural at a slightly faster rate, such as `1.2`. From bbdf0a4550447edd73021762a3bd79605c94e1bd Mon Sep 17 00:00:00 2001 From: Him188 Date: Mon, 28 Sep 2026 18:01:21 +0900 Subject: [PATCH 2/3] agents: enable strict language only when other languages must not be understood Co-Authored-By: Claude Opus 5.5 --- agents/build/voice-language.mdx | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/agents/build/voice-language.mdx b/agents/build/voice-language.mdx index 42cedbe..b8dc787 100644 --- a/agents/build/voice-language.mdx +++ b/agents/build/voice-language.mdx @@ -178,7 +178,7 @@ curl --request PATCH "https://api.fish.audio/v1/agent/agents/$AGENT_ID/config" \ - Language is detected per utterance, so a short or heavily accented phrase in the speaking language can occasionally be detected as another language and replaced with `[unintelligible speech]` too. Turn strict language on only when callers are expected to speak the speaking language. + Language is detected per utterance, so a short or heavily accented phrase in the speaking language can occasionally be detected as another language and replaced with `[unintelligible speech]` too. Only turn on strict language when you explicitly need to stop the agent from understanding other languages. ## Speaking speed From 4d2cc39f8e670eb4b82c04da6bb73649c64f8b05 Mon Sep 17 00:00:00 2001 From: Him188 Date: Mon, 28 Sep 2026 18:25:06 +0900 Subject: [PATCH 3/3] api-reference: refresh OpenAPI schema from the live API Co-Authored-By: Claude Opus 5.5 --- api-reference/openapi.json | 160 ++++++++++++++++++++++++++++++------- 1 file changed, 130 insertions(+), 30 deletions(-) diff --git a/api-reference/openapi.json b/api-reference/openapi.json index 1d9e04f..753f385 100644 --- a/api-reference/openapi.json +++ b/api-reference/openapi.json @@ -3613,7 +3613,10 @@ "$ref": "#/components/schemas/AgentPromptConfig" }, "voice": { - "$ref": "#/components/schemas/AgentVoiceConfig" + "$ref": "#/components/schemas/AgentVoiceConfigView" + }, + "asr": { + "$ref": "#/components/schemas/AgentAsrConfig" }, "conversation": { "$ref": "#/components/schemas/AgentConversationConfig" @@ -3647,6 +3650,7 @@ "config_hash", "prompt", "voice", + "asr", "conversation", "tools", "webhooks", @@ -3779,7 +3783,7 @@ }, "patch": { "summary": "Update Draft Config", - "description": "Patch the draft configuration section by section; omitted sections keep\ntheir value. Changes only affect live sessions after the next publish.\n`prompt.system_prompt` is limited to 32000 tokens (422 beyond); keeping it\nunder 2000 tokens is recommended for latency and cost.\n`voice.voice_id` accepts any public voice model id.\n`voice.speaking_language` accepts any of the 52 supported ISO 639-1 codes\n(the same set the console offers, see the Voice & language docs); anything else is 422. `voice.expressive` (default `true`) enables richer\nexpressive delivery (emotion steering, laughter and sounds, pauses); off\nkeeps the standard delivery. `voice.keyterms` is a speech-recognition vocabulary of\nup to 50 plain terms (brand names, product terms, personal names), each at\nmost 100 characters with no commas or semicolons; `[]` clears it and 20-50\nfocused terms work best. `tool_ids` and\n`knowledge_source_ids` replace their attachment lists wholesale and every\nid must resolve, else 422. `llm.custom` points the agent at your own\nOpenAI-compatible endpoint; mutually exclusive with `llm.model`, cleared\nwith an explicit null.", + "description": "Patch the draft configuration section by section; omitted sections keep\ntheir value. Changes only affect live sessions after the next publish.\n`prompt.system_prompt` is limited to 32000 tokens (422 beyond); keeping it\nunder 2000 tokens is recommended for latency and cost.\n`voice.voice_id` accepts any public voice model id.\n`voice.speaking_language` accepts any of the 52 supported ISO 639-1 codes\n(the same set the console offers, see the Voice & language docs); anything else is 422. `voice.expressive` (default `true`) enables richer\nexpressive delivery (emotion steering, laughter and sounds, pauses); off\nkeeps the standard delivery. The `asr` section configures speech\nrecognition. `asr.model` picks the model: `deepgram:nova-3` (default),\n`elevenlabs:scribe_v2_realtime`, or `elevenlabs:scribe_v2_medical`.\n`asr.multilingual` (default `false`) lets recognition follow callers who\nswitch language mid-call, off tells the recognizer to expect\n`voice.speaking_language` (models other than `deepgram:nova-3` may still\ntranscribe clear speech in another language). `asr.strict_language`\n(default `false`) enforces `voice.speaking_language` whatever\n`asr.multilingual` says: speech recognized as another language reaches the\nagent as `[unintelligible speech]`. Language is detected per utterance, so\na short or heavily accented phrase in the speaking language can\noccasionally be detected as another language and replaced too. Only\nenable it when you explicitly need to stop the agent from understanding\nother languages.\n`asr.keyterms` is a recognition\nvocabulary of up to 50 plain terms (brand names, product terms, personal\nnames), each at most 100 characters with no commas or semicolons. `[]`\nclears it and 20-50 focused terms work best. `tool_ids` and\n`knowledge_source_ids` replace their attachment lists wholesale and every\nid must resolve, else 422. `llm.custom` points the agent at your own\nOpenAI-compatible endpoint; mutually exclusive with `llm.model`, cleared\nwith an explicit null.", "security": [ { "BearerAuth": [] @@ -3829,7 +3833,10 @@ "$ref": "#/components/schemas/AgentPromptConfig" }, "voice": { - "$ref": "#/components/schemas/AgentVoiceConfig" + "$ref": "#/components/schemas/AgentVoiceConfigView" + }, + "asr": { + "$ref": "#/components/schemas/AgentAsrConfig" }, "conversation": { "$ref": "#/components/schemas/AgentConversationConfig" @@ -3863,6 +3870,7 @@ "config_hash", "prompt", "voice", + "asr", "conversation", "tools", "webhooks", @@ -16022,6 +16030,71 @@ "title": "PublicAgentAnalysisSummaryPatch", "type": "object" }, + "PublicAgentAsrPatch": { + "additionalProperties": false, + "properties": { + "model": { + "anyOf": [ + { + "enum": [ + "deepgram:nova-3", + "elevenlabs:scribe_v2_realtime", + "elevenlabs:scribe_v2_medical" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Model" + }, + "multilingual": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Multilingual" + }, + "strict_language": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "default": null, + "description": "Enforces voice.speaking_language whatever multilingual says: speech recognized as another language reaches the agent as [unintelligible speech]. Language is detected per utterance, so a short or heavily accented phrase in the speaking language can occasionally be detected as another language and replaced too. With deepgram:nova-3, speech in another language is usually not transcribed at all. Only enable it when you explicitly need to stop the agent from understanding other languages.", + "title": "Strict Language" + }, + "keyterms": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "maxItems": 50, + "type": "array" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Keyterms" + } + }, + "title": "PublicAgentAsrPatch", + "type": "object" + }, "PublicAgentConfigPatchPayload": { "additionalProperties": false, "properties": { @@ -16047,6 +16120,17 @@ ], "default": null }, + "asr": { + "anyOf": [ + { + "$ref": "#/components/schemas/PublicAgentAsrPatch" + }, + { + "type": "null" + } + ], + "default": null + }, "conversation": { "anyOf": [ { @@ -16725,22 +16809,6 @@ "default": null, "title": "Expressive" }, - "keyterms": { - "anyOf": [ - { - "items": { - "type": "string" - }, - "maxItems": 50, - "type": "array" - }, - { - "type": "null" - } - ], - "default": null, - "title": "Keyterms" - }, "speed": { "anyOf": [ { @@ -17158,6 +17226,41 @@ "title": "AgentAnalysisSummaryConfig", "type": "object" }, + "AgentAsrConfig": { + "properties": { + "model": { + "default": "deepgram:nova-3", + "enum": [ + "deepgram:nova-3", + "elevenlabs:scribe_v2_realtime", + "elevenlabs:scribe_v2_medical" + ], + "title": "Model", + "type": "string" + }, + "multilingual": { + "default": false, + "title": "Multilingual", + "type": "boolean" + }, + "strict_language": { + "default": false, + "description": "Enforces voice.speaking_language whatever multilingual says: speech recognized as another language reaches the agent as [unintelligible speech]. Language is detected per utterance, so a short or heavily accented phrase in the speaking language can occasionally be detected as another language and replaced too. With deepgram:nova-3, speech in another language is usually not transcribed at all. Only enable it when you explicitly need to stop the agent from understanding other languages.", + "title": "Strict Language", + "type": "boolean" + }, + "keyterms": { + "items": { + "type": "string" + }, + "maxItems": 50, + "title": "Keyterms", + "type": "array" + } + }, + "title": "AgentAsrConfig", + "type": "object" + }, "AgentConversationConfig": { "properties": { "max_duration_seconds": { @@ -17526,7 +17629,8 @@ "title": "AgentTransferOnFailure", "type": "object" }, - "AgentVoiceConfig": { + "AgentVoiceConfigView": { + "description": "Read shape of the voice section. The ASR fields that moved to the asr\nsection stay mirrored here, deprecated, so clients written before the split\nkeep reading them.", "properties": { "voice_id": { "default": "b347db033a6549378b48d00acb0d06cd", @@ -17597,14 +17701,6 @@ "title": "Expressive", "type": "boolean" }, - "keyterms": { - "items": { - "type": "string" - }, - "maxItems": 50, - "title": "Keyterms", - "type": "array" - }, "speed": { "default": 1, "maximum": 2, @@ -17613,7 +17709,7 @@ "type": "number" } }, - "title": "AgentVoiceConfig", + "title": "AgentVoiceConfigView", "type": "object" }, "PublicAgentKnowledgeBaseConfig": { @@ -17785,7 +17881,10 @@ "$ref": "#/components/schemas/AgentPromptConfig" }, "voice": { - "$ref": "#/components/schemas/AgentVoiceConfig" + "$ref": "#/components/schemas/AgentVoiceConfigView" + }, + "asr": { + "$ref": "#/components/schemas/AgentAsrConfig" }, "conversation": { "$ref": "#/components/schemas/AgentConversationConfig" @@ -17819,6 +17918,7 @@ "config_hash", "prompt", "voice", + "asr", "conversation", "tools", "webhooks",