Retell AI Create Phone Call API
The Create Phone Call API from Retell AI — 1 operation(s) for create phone call.
The Create Phone Call API from Retell AI — 1 operation(s) for create phone call.
openapi: 3.0.3
info:
title: Retell SDK Add Community Voice Create Phone Call API
version: 3.0.0
contact:
name: Retell Support
url: https://www.retellai.com/
email: support@retellai.com
license:
name: Apache 2.0
url: https://www.apache.org/licenses/LICENSE-2.0.html
servers:
- url: https://api.retellai.com
description: The production server.
security:
- api_key: []
tags:
- name: Create Phone Call
paths:
/v2/create-phone-call:
post:
description: Create a new outbound phone call
operationId: createPhoneCall
requestBody:
required: true
content:
application/json:
schema:
type: object
required:
- from_number
- to_number
properties:
from_number:
type: string
minLength: 1
example: '+14157774444'
description: The number you own in E.164 format. Must be a number purchased from Retell or imported to Retell.
to_number:
type: string
minLength: 1
example: '+12137774445'
description: The number you want to call, in E.164 format. If using a number purchased from Retell, only US numbers are supported as destination.
override_agent_id:
type: string
minLength: 1
example: oBeDLoLOeuAbiuaMFXRtDOLriTJ5tSxD
description: For this particular call, override the agent used with this agent id. This does not bind the agent to this number, this is for one time override.
override_agent_version:
$ref: '#/components/schemas/AgentVersionReference'
description: For this particular call, override the agent version used with this version. This does not bind the agent version to this number, this is for one time override.
agent_override:
$ref: '#/components/schemas/AgentOverrideRequest'
description: For this particular call, override agent configuration with these settings. This allows you to customize agent behavior for individual calls without modifying the base agent.
metadata:
type: object
description: An arbitrary object for storage purpose only. You can put anything here like your internal customer id associated with the call. Not used for processing. You can later get this field from the call object.
retell_llm_dynamic_variables:
type: object
additionalProperties: {}
example:
customer_name: John Doe
description: Add optional dynamic variables in key value pairs of string that injects into your Response Engine prompt and tool description. Only applicable for Response Engine.
custom_sip_headers:
type: object
additionalProperties:
type: string
example:
X-Custom-Header: Custom Value
description: Add optional custom SIP headers to the call.
ignore_e164_validation:
type: boolean
description: If true, the e.164 validation will be ignored for the from_number. This can be useful when you want to dial to internal pseudo numbers. This only applies when you are using custom telephony and does not apply when you are using Retell Telephony. If omitted, the default value is false.
example: true
responses:
'201':
description: Successfully made an outbound call.
content:
application/json:
schema:
$ref: '#/components/schemas/V2PhoneCallResponse'
'400':
$ref: '#/components/responses/BadRequest'
'401':
$ref: '#/components/responses/Unauthorized'
'402':
$ref: '#/components/responses/PaymentRequired'
'422':
$ref: '#/components/responses/UnprocessableContent'
'429':
$ref: '#/components/responses/TooManyRequests'
'500':
$ref: '#/components/responses/InternalServerError'
tags:
- Create Phone Call
components:
schemas:
CallLatency:
type: object
properties:
p50:
type: number
description: 50 percentile of latency, measured in milliseconds.
example: 800
p90:
type: number
description: 90 percentile of latency, measured in milliseconds.
example: 1200
p95:
type: number
description: 95 percentile of latency, measured in milliseconds.
example: 1500
p99:
type: number
description: 99 percentile of latency, measured in milliseconds.
example: 2500
max:
type: number
description: Maximum latency in the call, measured in milliseconds.
example: 2700
min:
type: number
description: Minimum latency in the call, measured in milliseconds.
example: 500
num:
type: number
description: Number of data points (number of times latency is tracked).
example: 10
values:
type: array
items:
type: number
description: All the latency data points in the call, measured in milliseconds.
ToolCallResultUtterance:
type: object
required:
- role
- tool_call_id
- content
properties:
role:
type: string
enum:
- tool_call_result
description: This is the result of a tool call.
tool_call_id:
type: string
description: Tool call id, globally unique.
content:
type: string
description: Result of the tool call, can be a string, a stringified json, etc.
successful:
type: boolean
description: Whether the tool call was successful.
CallPresetAnalysisData:
type: object
required:
- type
- name
description: System preset for post-call analysis (voice agents). Use in post_call_analysis_data to override prompts or mark fields optional.
properties:
type:
type: string
enum:
- system-presets
description: Identifies this item as a system preset.
name:
type: string
enum:
- call_summary
- call_successful
- user_sentiment
description: Preset identifier for voice agent analysis.
example: call_summary
description:
type: string
minLength: 1
description: Prompt or description for this preset.
required:
type: boolean
description: If false, this field is optional in the analysis. If true or unset, the field is required.
conditional_prompt:
type: string
description: Optional instruction to help decide whether this field needs to be populated. If not set, the field is always included.
CallAnalysis:
type: object
properties:
call_summary:
type: string
example: The agent called the user to ask question about his purchase inquiry. The agent asked several questions regarding his preference and asked if user would like to book an appointment. The user happily agreed and scheduled an appointment next Monday 10am.
description: A high level summary of the call.
in_voicemail:
type: boolean
example: false
description: Whether the call is entered voicemail.
user_sentiment:
type: string
enum:
- Negative
- Positive
- Neutral
- Unknown
example: Positive
description: Sentiment of the user in the call.
call_successful:
type: boolean
example: true
description: Whether the agent seems to have a successful call with the user, where the agent finishes the task, and the call was complete without being cutoff.
custom_analysis_data:
type: object
description: Custom analysis data that was extracted based on the schema defined in agent post call analysis data. Can be empty if nothing is specified.
VoicemailActionBridgeTransfer:
type: object
properties:
type:
type: string
enum:
- bridge_transfer
example: bridge_transfer
required:
- type
UtteranceOrToolCall:
oneOf:
- $ref: '#/components/schemas/Utterance'
- $ref: '#/components/schemas/ToolCallInvocationUtterance'
- $ref: '#/components/schemas/ToolCallResultUtterance'
- $ref: '#/components/schemas/NodeTransitionUtterance'
- $ref: '#/components/schemas/DTMFUtterance'
VoicemailActionPrompt:
type: object
properties:
type:
type: string
enum:
- prompt
example: prompt
text:
type: string
example: Summarize the call in 2 sentences.
description: The prompt used to generate the text to be spoken when the call is detected to be in voicemail.
required:
- type
- text
NodeTransitionUtterance:
type: object
required:
- role
- former_node_id
- former_node_name
- new_node_id
- new_node_name
properties:
role:
type: string
enum:
- node_transition
description: This is result of a node transition
former_node_id:
type: string
description: Former node id
former_node_name:
type: string
description: Former node name
new_node_id:
type: string
description: New node id
new_node_name:
type: string
description: New node name
transition_type:
type: string
enum:
- global
- global_go_back
- interrupt_go_back
- normal
description: How this node was reached. "global" means a global node transition, "global_go_back" means returning from a global node, "interrupt_go_back" means going back due to user interruption, and "normal" means a regular edge transition.
LanguageLegacy:
oneOf:
- $ref: '#/components/schemas/Language'
- type: string
enum:
- multi
example: en-US
description: Legacy single-string language format. Accepts any concrete locale from `Language`, plus the special scalar value `multi` for multilingual support. If unset, will use default value `en-US`.
Language:
type: string
example: en-US
enum:
- en-US
- en-IN
- en-GB
- en-AU
- en-NZ
- de-DE
- es-ES
- es-419
- hi-IN
- fr-FR
- fr-CA
- ja-JP
- pt-PT
- pt-BR
- zh-CN
- ru-RU
- it-IT
- ko-KR
- nl-NL
- nl-BE
- pl-PL
- tr-TR
- vi-VN
- ro-RO
- bg-BG
- ca-ES
- th-TH
- da-DK
- fi-FI
- el-GR
- hu-HU
- id-ID
- no-NO
- sk-SK
- sv-SE
- lt-LT
- lv-LV
- cs-CZ
- ms-MY
- af-ZA
- ar-SA
- az-AZ
- bs-BA
- cy-GB
- fa-IR
- fil-PH
- gl-ES
- he-IL
- hr-HR
- hy-AM
- is-IS
- kk-KZ
- kn-IN
- mk-MK
- mr-IN
- ne-NP
- sl-SI
- sr-RS
- sw-KE
- ta-IN
- ur-IN
- yue-CN
- uk-UA
description: Specifies what language (and dialect) the agent will operate in. For instance, selecting `en-GB` optimizes speech recognition for British English and indexes knowledge bases with English. If unset, will use default value `en-US`. This enum does not include the legacy scalar value `multi`.
ConversationFlowOverride:
type: object
description: Override properties for conversation flow configuration in agent override requests.
properties:
model_choice:
$ref: '#/components/schemas/ModelChoice'
description: The model choice for the conversation flow.
model_temperature:
type: number
minimum: 0
maximum: 1
example: 0.7
description: Controls the randomness of the model's responses. Lower values make responses more deterministic.
nullable: true
tool_call_strict_mode:
type: boolean
example: true
description: Whether to use strict mode for tool calls. Only applicable when using certain supported models.
nullable: true
knowledge_base_ids:
type: array
items:
type: string
example:
- kb_001
- kb_002
description: Knowledge base IDs for RAG (Retrieval-Augmented Generation).
nullable: true
kb_config:
type: object
$ref: '#/components/schemas/KBConfig'
description: Knowledge base configuration for RAG retrieval.
start_speaker:
type: string
enum:
- user
- agent
example: agent
description: Who starts the conversation - user or agent.
begin_after_user_silence_ms:
type: integer
example: 2000
description: If set, the AI will begin the conversation after waiting for the user for the duration (in milliseconds) specified by this attribute. This only applies if the agent is configured to wait for the user to speak first. If not set, the agent will wait indefinitely for the user to speak.
nullable: true
ModelChoice:
oneOf:
- $ref: '#/components/schemas/ModelChoiceCascading'
ProductCost:
type: object
required:
- product
- cost
properties:
product:
type: string
description: Product name that has a cost associated with it.
example: elevenlabs_tts
unit_price:
type: number
description: Unit price of the product in cents per second.
example: 1
cost:
type: number
description: Cost for the product in cents for the duration of the call.
example: 60
is_transfer_leg_cost:
type: boolean
description: True if this cost item is for a transfer segment.
AgentRequest:
type: object
properties:
response_engine:
$ref: '#/components/schemas/ResponseEngine'
description: The Response Engine to attach to the agent. It is used to generate responses for the agent. You need to create a Response Engine first before attaching it to an agent.
example:
type: retell-llm
llm_id: llm_234sdertfsdsfsdf
version: 0
agent_name:
type: string
example: Jarvis
description: The name of the agent. Only used for your own reference.
nullable: true
version_description:
type: string
example: Customer support agent for handling product inquiries
description: Optional description of the agent version. Used for your own reference and documentation.
nullable: true
voice_id:
type: string
example: retell-Cimo
description: Unique voice id used for the agent. Find list of available voices and their preview in Dashboard.
voice_model:
type: string
enum:
- eleven_turbo_v2
- eleven_flash_v2
- eleven_turbo_v2_5
- eleven_flash_v2_5
- eleven_multilingual_v2
- eleven_v3
- sonic-3
- sonic-3-latest
- tts-1
- gpt-4o-mini-tts
- speech-02-turbo
- speech-2.8-turbo
- s1
- s2-pro
- null
description: Select the voice model used for the selected voice. Each provider has a set of available voice models. Set to null to remove voice model selection, and default ones will apply. Check out dashboard for more details of each voice model.
nullable: true
fallback_voice_ids:
type: array
items:
type: string
example:
- cartesia-Cimo
- minimax-Cimo
description: When TTS provider for the selected voice is experiencing outages, we would use fallback voices listed here for the agent. Voice id and the fallback voice ids must be from different TTS providers. The system would go through the list in order, if the first one in the list is also having outage, it would use the next one. Set to null to remove voice fallback for the agent.
nullable: true
voice_temperature:
type: number
example: 1
description: Controls how stable the voice is. Value ranging from [0,2]. Lower value means more stable, and higher value means more variant speech generation. Check the dashboard to see what provider supports this feature. If unset, default value 1 will apply.
voice_speed:
type: number
minimum: 0.5
maximum: 2
example: 1
description: Controls speed of voice. Value ranging from [0.5,2]. Lower value means slower speech, while higher value means faster speech rate. If unset, default value 1 will apply.
enable_dynamic_voice_speed:
type: boolean
example: true
description: If set to true, will enable dynamic voice speed adjustment based on the user's speech rate and conversation context. If unset, default value false will apply.
enable_dynamic_responsiveness:
type: boolean
example: true
description: If set to true, the agent will dynamically adjust how quickly it responds based on the user's speech rate and past turn-taking behavior in the call. If unset, default value false will apply.
volume:
type: number
example: 1
description: If set, will control the volume of the agent. Value ranging from [0,2]. Lower value means quieter agent speech, while higher value means louder agent speech. If unset, default value 1 will apply.
voice_emotion:
type: string
nullable: true
enum:
- calm
- sympathetic
- happy
- sad
- angry
- fearful
- surprised
- null
example: calm
description: 'Controls the emotional tone of the agent''s voice. Currently supported for Cartesia and Minimax TTS providers. If unset, no emotion will be used.
'
responsiveness:
type: number
minimum: 0
maximum: 1
example: 1
description: Controls how responsive is the agent. Value ranging from [0,1]. Lower value means less responsive agent (wait more, respond slower), while higher value means faster exchanges (respond when it can). If unset, default value 1 will apply.
interruption_sensitivity:
type: number
minimum: 0
maximum: 1
example: 1
description: Controls how sensitive the agent is to user interruptions. Value ranging from [0,1]. Lower value means it will take longer / more words for user to interrupt agent, while higher value means it's easier for user to interrupt agent. If unset, default value 1 will apply. When this is set to 0, agent would never be interrupted.
enable_backchannel:
type: boolean
example: true
description: Controls whether the agent would backchannel (agent interjects the speaker with phrases like "yeah", "uh-huh" to signify interest and engagement). Backchannel when enabled tends to show up more in longer user utterances. If not set, agent will not backchannel.
backchannel_frequency:
type: number
example: 0.9
description: Only applicable when enable_backchannel is true. Controls how often the agent would backchannel when a backchannel is possible. Value ranging from [0,1]. Lower value means less frequent backchannel, while higher value means more frequent backchannel. If unset, default value 0.8 will apply.
backchannel_words:
type: array
items:
type: string
example:
- yeah
- uh-huh
description: Only applicable when enable_backchannel is true. A list of words that the agent would use as backchannel. If not set, default backchannel words will apply. Check out [backchannel default words](/agent/interaction-configuration#backchannel) for more details. Note that certain voices do not work too well with certain words, so it's recommended to experiment before adding any words.
nullable: true
reminder_trigger_ms:
type: number
example: 10000
description: If set (in milliseconds), will trigger a reminder to the agent to speak if the user has been silent for the specified duration after some agent speech. Must be a positive number. If unset, default value of 10000 ms (10 s) will apply.
reminder_max_count:
type: integer
example: 2
description: If set, controls how many times agent would remind user when user is unresponsive. Must be a non negative integer. If unset, default value of 1 will apply (remind once). Set to 0 to disable agent from reminding.
ambient_sound:
type: string
enum:
- coffee-shop
- convention-hall
- summer-outdoor
- mountain-outdoor
- static-noise
- call-center
- null
description: 'If set, will add ambient environment sound to the call to make experience more realistic. Currently supports the following options:
- `coffee-shop`: Coffee shop ambience with people chatting in background. [Listen to Ambience](https://retell-utils-public.s3.us-west-2.amazonaws.com/coffee-shop.wav)
- `convention-hall`: Convention hall ambience, with some echo and people chatting in background. [Listen to Ambience](https://retell-utils-public.s3.us-west-2.amazonaws.com/convention-hall.wav)
- `summer-outdoor`: Summer outdoor ambience with cicada chirping. [Listen to Ambience](https://retell-utils-public.s3.us-west-2.amazonaws.com/summer-outdoor.wav)
- `mountain-outdoor`: Mountain outdoor ambience with birds singing. [Listen to Ambience](https://retell-utils-public.s3.us-west-2.amazonaws.com/mountain-outdoor.wav)
- `static-noise`: Constant static noise. [Listen to Ambience](https://retell-utils-public.s3.us-west-2.amazonaws.com/static-noise.wav)
- `call-center`: Call center work noise. [Listen to Ambience](https://retell-utils-public.s3.us-west-2.amazonaws.com/call-center.wav)
Set to `null` to remove ambient sound from this agent.
'
nullable: true
ambient_sound_volume:
type: number
example: 1
description: If set, will control the volume of the ambient sound. Value ranging from [0,2]. Lower value means quieter ambient sound, while higher value means louder ambient sound. If unset, default value 1 will apply.
language:
oneOf:
- $ref: '#/components/schemas/LanguageLegacy'
- type: array
items:
$ref: '#/components/schemas/Language'
description: Specifies what language(s) the agent will operate in. Accepts either a single scalar locale (e.g. `en-US`), the legacy scalar value `multi` for multilingual support, or an array of concrete locale codes for explicit multi-locale selection (e.g. `["en-US","es-ES"]`). The array form must contain concrete locale codes only — the `multi` value is valid only as the scalar legacy form and must not appear inside an array. Single-element arrays are normalized to the equivalent scalar on output. If unset, defaults to `en-US`.
webhook_url:
type: string
example: https://webhook-url-here
description: The webhook for agent to listen to call events. See what events it would get at [webhook doc](/features/webhook). If set, will binds webhook events for this agent to the specified url, and will ignore the account level webhook for this agent. Set to `null` to remove webhook url from this agent.
nullable: true
webhook_events:
type: array
items:
type: string
enum:
- call_started
- call_ended
- call_analyzed
- transcript_updated
- transfer_started
- transfer_bridged
- transfer_cancelled
- transfer_ended
description: Which webhook events this agent should receive. If not set, defaults to call_started, call_ended, call_analyzed.
nullable: true
webhook_timeout_ms:
type: integer
example: 10000
description: The timeout for the webhook in milliseconds. If not set, default value of 10000 will apply.
boosted_keywords:
type: array
items:
type: string
example:
- retell
- kroger
description: Provide a customized list of keywords to bias the transcriber model, so that these words are more likely to get transcribed. Commonly used for names, brands, street, etc.
nullable: true
data_storage_setting:
type: string
enum:
- everything
- everything_except_pii
- basic_attributes_only
example: everything
description: 'Granular setting to manage how Retell stores sensitive data (transcripts, recordings, logs, etc.).
This replaces the deprecated `opt_out_sensitive_data_storage` field.
- `everything`: Store all data including transcripts, recordings, and logs.
- `everything_except_pii`: Store data without PII when PII is detected.
- `basic_attributes_only`: Store only basic attributes; no transcripts/recordings/logs.
If not set, default value of "everything" will apply.
'
data_storage_retention_days:
type: integer
minimum: 1
maximum: 730
example: 30
nullable: true
description: Number of days to retain call/chat data before automatic deletion. Must be between 1 and 730 days. If not set, data is retained forever (no automatic deletion).
opt_in_signed_url:
type: boolean
example: true
description: Whether this agent opts in for signed URLs for public logs and recordings. When enabled, the generated URLs will include security signatures that restrict access and automatically expire after 24 hours.
signed_url_expiration_ms:
type: integer
example: 86400000
description: The expiration time for the signed url in milliseconds. Only applicable when opt_in_signed_url is true. If not set, default value of 86400000 (24 hours) will apply.
nullable: true
pronunciation_dictionary:
type: array
items:
type: object
required:
- word
- alphabet
- phoneme
properties:
word:
type: string
example: actually
description: The string of word / phrase to be annotated with pronunciation.
alphabet:
type: string
enum:
- ipa
- cmu
example: ipa
description: The phonetic alphabet to be used for pronunciation.
phoneme:
type: string
example: ˈæktʃuəli
description: Pronunciation of the word in the format of a IPA / CMU pronunciation.
description: A list of words / phrases and their pronunciation to be used to guide the audio synthesize for consistent pronunciation. Check the dashboard to see what provider supports this feature. Set to null to remove pronunciation dictionary from this agent.
nullable: true
end_call_after_silence_ms:
type: integer
example: 600000
description: If users stay silent for a period after agent speech, end the call. The minimum value allowed is 10,000 ms (10 s). By default, this is set to 600000 (10 min).
max_call_duration_ms:
type: integer
example: 3600000
description: Maximum allowed length for the call, will force end the call if reached. The minimum value allowed is 60,000 ms (1 min), and maximum value allowed is 7,200,000 (2 hours). By default, this is set to 3,600,000 (1 hour).
voicemail_message:
type: string
example: Hi, please give us a callback.
description: The message to be played when the call enters a voicemail. Note that this feature is only available for phone calls. If you want to hangup after hitting voicemail, set this to empty string.
voicemail_detection_timeout_ms:
type: integer
example: 30000
description: Configures when to stop running voicemail detection, as it becomes unlikely to hit voicemail after a couple minutes, and keep running it will only have negative impact. The minimum value allowed is 5,000 ms (5 s), and maximum value allowed is 180,000 (3 minutes). By default, this is set to 30,000 (30 s).
voicemail_option:
type: object
properties:
action:
$ref: '#/components/schemas/VoicemailAction'
detection_prompt:
type: string
maxLength: 2000
nullable: true
description: Optionally describe what should be treated as voicemail. Leave as null to use the default definition.
required:
- action
description: If this option is set, the call will try to detect voicemail in the first 3 minutes of the call. Actions defined (hangup, or leave a message) will be applied when the voicemail is detected. Set this to null to disable voicemail detection.
example:
action:
type: static_text
text: Please give us a callback tomorrow at 10am.
nullable: true
ivr_option:
type: object
properties:
action:
$ref: '#/components/schemas/IvrAction'
detection_prompt:
type: string
maxLength: 2000
nullable: true
description: Optionally describe what should be treated as an IVR. Leave as null to use the default definition.
required:
- action
description: If this option is set, the call will try to detect IVR in the first 3 minutes of the call. Actions defined will be applied when the IVR is detected. Set this to null to disable IVR detection.
example:
action:
type: hangup
nullable: true
call_screening_option:
$ref: '#/components/schemas/CallScreeningOption'
post_call_analysis_data:
type: array
items:
$ref: '#/components/schemas/PostCallAnalysisData'
description: Post call analysis data to extract from the call. This data will augment the pre-defined variables extracted in the call analysis. This will be available after the call ends.
nullable: true
post_call_analysis_model:
$ref: '#/components/schemas/NullableLLMModel'
example: gpt-4.1-mini
description: The model to use for post call analysis. Default to gpt-4.1.
analysis_successful_prompt:
type: string
maxLength: 2000
example: The agent finished the task and the call was complete without being cutoff.
description: Prompt to determine whether the post call or chat analysis should mark the interaction as successful. Set to null to use the default prompt.
nullable: true
analysis_summary_prompt:
type: string
maxLength: 2000
example: Summarize the out
# --- truncated at 32 KB (75 KB total) ---
# Full source: https://raw.githubusercontent.com/api-evangelist/retell-ai/refs/heads/main/openapi/retell-ai-create-phone-call-api-openapi.yml