{"asyncapi":"2.6.0","info":{"title":"Speech-to-speech (EVI)","version":"1.0.0"},"channels":{"/chat":{"description":"Chat with Empathic Voice Interface (EVI)","bindings":{"ws":{"query":{"type":"object","properties":{"access_token":{"type":"string","default":""},"allow_connection":{"type":"boolean","default":false},"config_id":{"type":"string"},"config_version":{"type":"integer"},"event_limit":{"type":"integer"},"resumed_chat_group_id":{"type":"string"},"verbose_transcription":{"type":"boolean","default":false},"api_key":{"type":"string","default":""},"session_settings":{"$ref":"#/components/schemas/_chat_session_settings"}}}}},"publish":{"operationId":"subpackage_chat.chat-publish","summary":"subscribe","message":{"name":"subscribe","title":"subscribe","payload":{"$ref":"#/components/schemas/ChatSubscribe"}}},"subscribe":{"operationId":"subpackage_chat.chat-subscribe","summary":"publish","message":{"name":"publish","title":"publish","payload":{"$ref":"#/components/schemas/ChatPublish"}}}},"/chat/{chat_id}/connect":{"description":"Connects to an in-progress EVI chat session. The original chat must have been started with `allow_connection=true`. The connection can be used to send and receive the same messages as the original chat, with the exception that `audio_input` messages are not allowed.","parameters":{"chat_id":{"description":"The ID of the chat to connect to.","schema":{"type":"string"}}},"bindings":{"ws":{"query":{"type":"object","properties":{"access_token":{"type":"string","default":""}}}}},"publish":{"operationId":"subpackage_controlPlane.chatChatIdConnect-publish","summary":"subscribe","message":{"name":"subscribe","title":"subscribe","payload":{"$ref":"#/components/schemas/ChatChatIdConnectSubscribe"}}},"subscribe":{"operationId":"subpackage_controlPlane.chatChatIdConnect-subscribe","summary":"publish","message":{"name":"publish","title":"publish","payload":{"$ref":"#/components/schemas/ChatChatIdConnectPublish"}}}}},"servers":{"prod":{"url":"wss://api.hume.ai/v0/evi","protocol":"wss","x-default":true}},"components":{"schemas":{"Encoding":{"type":"string","enum":["linear16"],"title":"Encoding"},"ChannelsChatSessionSettingsSchemaAudio":{"type":"object","properties":{"channels":{"type":"integer","description":"Sets number of audio channels for audio input."},"encoding":{"$ref":"#/components/schemas/Encoding","description":"Sets encoding format of the audio input, such as `linear16`."},"sample_rate":{"type":"integer","description":"Sets the sample rate for audio input. (Number of samples per second in the audio input, measured in Hertz.)"}},"description":"Configuration details for the audio input used during the session. Ensures the audio is being correctly set up for processing.\n\nThis optional field is only required when the audio input is encoded in PCM Linear 16 (16-bit, little-endian, signed PCM WAV data). For detailed instructions on how to configure session settings for PCM Linear 16 audio, please refer to the [Session Settings section](/docs/empathic-voice-interface-evi/configuration#session-settings) on the EVI Configuration page.","title":"ChannelsChatSessionSettingsSchemaAudio"},"ContextType":{"type":"string","enum":["persistent","temporary"],"title":"ContextType"},"ChannelsChatSessionSettingsSchemaContext":{"type":"object","properties":{"text":{"type":"string","description":"The context to be injected into the conversation. Helps inform the LLM's response by providing relevant information about the ongoing conversation.\n\nThis text will be appended to the end of [user_messages](/reference/speech-to-speech-evi/chat#receive.UserMessage.message.content) based on the chosen persistence level. For example, if you want to remind EVI of its role as a helpful weather assistant, the context you insert will be appended to the end of user messages as `{Context: You are a helpful weather assistant}`."},"type":{"$ref":"#/components/schemas/ContextType","description":"The persistence level of the injected context. Specifies how long the injected context will remain active in the session.\n\n- **Temporary**: Context that is only applied to the following assistant response.\n\n- **Persistent**: Context that is applied to all subsequent assistant responses for the remainder of the Chat."}},"description":"Field for injecting additional context into the conversation, which is appended to the end of user messages for the session.\n\nWhen included in a Session Settings message, the provided context can be used to remind the LLM of its role in every user message, prevent it from forgetting important details, or add new relevant information to the conversation.\n\nSet to `null` to clear injected context.","title":"ChannelsChatSessionSettingsSchemaContext"},"_chat_session_settings":{"type":"object","properties":{"audio":{"$ref":"#/components/schemas/ChannelsChatSessionSettingsSchemaAudio","description":"Configuration details for the audio input used during the session. Ensures the audio is being correctly set up for processing.\n\nThis optional field is only required when the audio input is encoded in PCM Linear 16 (16-bit, little-endian, signed PCM WAV data). For detailed instructions on how to configure session settings for PCM Linear 16 audio, please refer to the [Session Settings section](/docs/empathic-voice-interface-evi/configuration#session-settings) on the EVI Configuration page."},"context":{"$ref":"#/components/schemas/ChannelsChatSessionSettingsSchemaContext","description":"Field for injecting additional context into the conversation, which is appended to the end of user messages for the session.\n\nWhen included in a Session Settings message, the provided context can be used to remind the LLM of its role in every user message, prevent it from forgetting important details, or add new relevant information to the conversation.\n\nSet to `null` to clear injected context."},"custom_session_id":{"type":"string","description":"Unique identifier for the session. Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions.\n\nIf included, the response sent from Hume to your backend will include this ID. This allows you to correlate frontend users with their incoming messages.\n\nIt is recommended to pass a `custom_session_id` if you are using a Custom Language Model. Please see our guide to [using a custom language model](/docs/empathic-voice-interface-evi/custom-language-model) with EVI to learn more."},"event_limit":{"type":"integer","description":"The maximum number of chat events to return from chat history. By default, the system returns up to 300 events (100 events per page × 3 pages). Set this parameter to a smaller value to limit the number of events returned."},"language_model_api_key":{"type":"string","description":"Third party API key for the supplemental language model.\n\nWhen provided, EVI will use this key instead of Hume's API key for the supplemental LLM. This allows you to bypass rate limits and utilize your own API key as needed."},"system_prompt":{"type":"string","description":"Instructions used to shape EVI's behavior, responses, and style for the session.\n\nWhen included in a Session Settings message, the provided Prompt overrides the existing one specified in the EVI configuration. If no Prompt was defined in the configuration, this Prompt will be the one used for the session.\n\nYou can use the Prompt to define a specific goal or role for EVI, specifying how it should act or what it should focus on during the conversation. For example, EVI can be instructed to act as a customer support representative, a fitness coach, or a travel advisor, each with its own set of behaviors and response styles.\n\nFor help writing a system prompt, see our [Prompting Guide](/docs/empathic-voice-interface-evi/prompting)."},"voice_id":{"type":"string","description":"Allows you to change the voice during an active chat. Updating the voice does not affect chat context or conversation history."},"variables":{"type":["object","null"],"additionalProperties":{"description":"Any type"},"default":null,"description":"This field allows you to assign values to dynamic variables referenced in your system prompt.\n\nEach key represents the variable name, and the corresponding value is the specific content you wish to assign to that variable within the session. While the values for variables can be strings, numbers, or booleans, the value will ultimately be converted to a string when injected into your system prompt.\n\nWhen used in query parameters, specify each variable using bracket notation: `session_settings[variables][key]=value`. For example: `session_settings[variables][name]=John&session_settings[variables][age]=30`.\n\nUsing this field, you can personalize responses based on session-specific details. For more guidance, see our [guide on using dynamic variables](/docs/speech-to-speech-evi/features/dynamic-variables)."}},"title":"/chat_session_settings"},"AssistantEnd":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"type":{"type":"string","enum":["assistant_end"],"description":"The type of message sent through the socket; for an Assistant End message, this must be `assistant_end`.\n\nThis message indicates the conclusion of the assistant's response, signaling that the assistant has finished speaking for the current conversational turn."}},"required":["type"],"description":"**Indicates the conclusion of the assistant's response**, signaling that the assistant has finished speaking for the current conversational turn.","title":"AssistantEnd"},"Role":{"type":"string","enum":["assistant","system","user","all","tool","context"],"title":"Role"},"ToolType":{"type":"string","enum":["builtin","function"],"title":"ToolType"},"ToolCallMessage":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"name":{"type":"string","description":"Name of the tool called."},"parameters":{"type":"string","description":"Parameters of the tool.\n\nThese parameters define the inputs needed for the tool's execution, including the expected data type and description for each input field. Structured as a stringified JSON schema, this format ensures the tool receives data in the expected format."},"response_required":{"type":"boolean","description":"Indicates whether a response to the tool call is required from the developer, either in the form of a [Tool Response message](/reference/empathic-voice-interface-evi/chat/chat#send.Tool%20Response%20Message.type) or a [Tool Error message](/reference/empathic-voice-interface-evi/chat/chat#send.Tool%20Error%20Message.type)."},"tool_call_id":{"type":"string","description":"The unique identifier for a specific tool call instance.\n\nThis ID is used to track the request and response of a particular tool invocation, ensuring that the correct response is linked to the appropriate request."},"tool_type":{"$ref":"#/components/schemas/ToolType","description":"Type of tool called. Either `builtin` for natively implemented tools, like web search, or `function` for user-defined tools."},"type":{"type":"string","enum":["tool_call"],"description":"The type of message sent through the socket; for a Tool Call message, this must be `tool_call`.\n\nThis message indicates that the supplemental LLM has detected a need to invoke the specified tool."}},"required":["name","parameters","response_required","tool_call_id","tool_type","type"],"description":"**Indicates that the supplemental LLM has detected a need to invoke the specified tool.** This message is only received for user-defined function tools.\n\nContains the tool name, parameters (as a stringified JSON schema), whether a response is required from the developer (either in the form of a `ToolResponseMessage` or a `ToolErrorMessage`), the unique tool call ID for tracking the request and response, and the tool type. See our [Tool Use Guide](/docs/speech-to-speech-evi/features/tool-use) for further details.","title":"ToolCallMessage"},"ToolResponseMessage":{"type":"object","properties":{"content":{"type":"string","description":"Return value of the tool call. Contains the output generated by the tool to pass back to EVI."},"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"tool_call_id":{"type":"string","description":"The unique identifier for a specific tool call instance.\n\nThis ID is used to track the request and response of a particular tool invocation, ensuring that the correct response is linked to the appropriate request. The specified `tool_call_id` must match the one received in the [Tool Call message](/reference/empathic-voice-interface-evi/chat/chat#receive.Tool%20Call%20Message.tool_call_id)."},"tool_name":{"type":["string","null"],"default":null},"tool_type":{"oneOf":[{"$ref":"#/components/schemas/ToolType"},{"type":"null"}],"default":null},"type":{"type":"string","enum":["tool_response"],"description":"The type of message sent through the socket; for a Tool Response message, this must be `tool_response`.\n\nUpon receiving a [Tool Call message](/reference/empathic-voice-interface-evi/chat/chat#receive.Tool%20Call%20Message.type) and successfully invoking the function, this message is sent to convey the result of the function call back to EVI."}},"required":["content","tool_call_id","type"],"description":"**Return value of the tool call.** Contains the output generated by the tool to pass back to EVI. Upon receiving a Tool Call message and successfully invoking the function, this message is sent to convey the result of the function call back to EVI.\n\nFor built-in tools implemented on the server, you will receive this message type rather than a `ToolCallMessage`. See our [Tool Use Guide](/docs/speech-to-speech-evi/features/tool-use) for further details.","title":"ToolResponseMessage"},"ErrorLevel":{"type":"string","enum":["warn"],"title":"ErrorLevel"},"ToolErrorMessage":{"type":"object","properties":{"code":{"type":["string","null"],"default":null,"description":"Error code. Identifies the type of error encountered."},"content":{"type":["string","null"],"default":null,"description":"Optional text passed to the supplemental LLM in place of the tool call result. The LLM then uses this text to generate a response back to the user, ensuring continuity in the conversation if the tool errors."},"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"error":{"type":"string","description":"Error message from the tool call, not exposed to the LLM or user."},"level":{"oneOf":[{"$ref":"#/components/schemas/ErrorLevel"},{"type":"null"}],"default":"warn","description":"Indicates the severity of an error; for a Tool Error message, this must be `warn` to signal an unexpected event."},"tool_call_id":{"type":"string","description":"The unique identifier for a specific tool call instance.\n\nThis ID is used to track the request and response of a particular tool invocation, ensuring that the Tool Error message is linked to the appropriate tool call request. The specified `tool_call_id` must match the one received in the [Tool Call message](/reference/empathic-voice-interface-evi/chat/chat#receive.Tool%20Call%20Message.type)."},"tool_type":{"oneOf":[{"$ref":"#/components/schemas/ToolType"},{"type":"null"}],"default":"function","description":"Type of tool called. Either `builtin` for natively implemented tools, like web search, or `function` for user-defined tools."},"type":{"type":"string","enum":["tool_error"],"description":"The type of message sent through the socket; for a Tool Error message, this must be `tool_error`.\n\nUpon receiving a [Tool Call message](/reference/empathic-voice-interface-evi/chat/chat#receive.Tool%20Call%20Message.type) and failing to invoke the function, this message is sent to notify EVI of the tool's failure."}},"required":["error","tool_call_id","type"],"description":"**Error message from the tool call**, not exposed to the LLM or user. Upon receiving a Tool Call message and failing to invoke the function, this message is sent to notify EVI of the tool's failure.\n\nFor built-in tools implemented on the server, you will receive this message type rather than a `ToolCallMessage` if the tool fails. See our [Tool Use Guide](/docs/speech-to-speech-evi/features/tool-use) for further details.","title":"ToolErrorMessage"},"ChatMessageToolResult":{"oneOf":[{"$ref":"#/components/schemas/ToolResponseMessage"},{"$ref":"#/components/schemas/ToolErrorMessage"}],"description":"Function call response from client.","title":"ChatMessageToolResult"},"ChatMessage":{"type":"object","properties":{"content":{"type":["string","null"],"default":null,"description":"Transcript of the message."},"role":{"$ref":"#/components/schemas/Role","description":"Role of who is providing the message."},"tool_call":{"oneOf":[{"$ref":"#/components/schemas/ToolCallMessage"},{"type":"null"}],"default":null,"description":"Function call name and arguments."},"tool_result":{"oneOf":[{"$ref":"#/components/schemas/ChatMessageToolResult"},{"type":"null"}],"default":null,"description":"Function call response from client."}},"required":["role"],"title":"ChatMessage"},"EmotionScores":{"type":"object","properties":{"Admiration":{"type":"number","format":"double"},"Adoration":{"type":"number","format":"double"},"Aesthetic Appreciation":{"type":"number","format":"double"},"Amusement":{"type":"number","format":"double"},"Anger":{"type":"number","format":"double"},"Anxiety":{"type":"number","format":"double"},"Awe":{"type":"number","format":"double"},"Awkwardness":{"type":"number","format":"double"},"Boredom":{"type":"number","format":"double"},"Calmness":{"type":"number","format":"double"},"Concentration":{"type":"number","format":"double"},"Confusion":{"type":"number","format":"double"},"Contemplation":{"type":"number","format":"double"},"Contempt":{"type":"number","format":"double"},"Contentment":{"type":"number","format":"double"},"Craving":{"type":"number","format":"double"},"Desire":{"type":"number","format":"double"},"Determination":{"type":"number","format":"double"},"Disappointment":{"type":"number","format":"double"},"Disgust":{"type":"number","format":"double"},"Distress":{"type":"number","format":"double"},"Doubt":{"type":"number","format":"double"},"Ecstasy":{"type":"number","format":"double"},"Embarrassment":{"type":"number","format":"double"},"Empathic Pain":{"type":"number","format":"double"},"Entrancement":{"type":"number","format":"double"},"Envy":{"type":"number","format":"double"},"Excitement":{"type":"number","format":"double"},"Fear":{"type":"number","format":"double"},"Guilt":{"type":"number","format":"double"},"Horror":{"type":"number","format":"double"},"Interest":{"type":"number","format":"double"},"Joy":{"type":"number","format":"double"},"Love":{"type":"number","format":"double"},"Nostalgia":{"type":"number","format":"double"},"Pain":{"type":"number","format":"double"},"Pride":{"type":"number","format":"double"},"Realization":{"type":"number","format":"double"},"Relief":{"type":"number","format":"double"},"Romance":{"type":"number","format":"double"},"Sadness":{"type":"number","format":"double"},"Satisfaction":{"type":"number","format":"double"},"Shame":{"type":"number","format":"double"},"Surprise (negative)":{"type":"number","format":"double"},"Surprise (positive)":{"type":"number","format":"double"},"Sympathy":{"type":"number","format":"double"},"Tiredness":{"type":"number","format":"double"},"Triumph":{"type":"number","format":"double"}},"required":["Admiration","Adoration","Aesthetic Appreciation","Amusement","Anger","Anxiety","Awe","Awkwardness","Boredom","Calmness","Concentration","Confusion","Contemplation","Contempt","Contentment","Craving","Desire","Determination","Disappointment","Disgust","Distress","Doubt","Ecstasy","Embarrassment","Empathic Pain","Entrancement","Envy","Excitement","Fear","Guilt","Horror","Interest","Joy","Love","Nostalgia","Pain","Pride","Realization","Relief","Romance","Sadness","Satisfaction","Shame","Surprise (negative)","Surprise (positive)","Sympathy","Tiredness","Triumph"],"title":"EmotionScores"},"ProsodyInference":{"type":"object","properties":{"scores":{"$ref":"#/components/schemas/EmotionScores","description":"The confidence scores for 48 emotions within the detected expression of an audio sample.\n\nScores typically range from 0 to 1, with higher values indicating a stronger confidence level in the measured attribute.\n\nSee our guide on [interpreting expression measures](/docs/speech-to-speech-evi/faq#what-do-evis-expression-labels-and-measures-mean) to learn more."}},"required":["scores"],"title":"ProsodyInference"},"Inference":{"type":"object","properties":{"prosody":{"oneOf":[{"$ref":"#/components/schemas/ProsodyInference"},{"type":"null"}],"description":"Prosody model inference results.\n\nEVI uses the prosody model to measure 48 emotions related to speech and vocal characteristics within a given expression."}},"required":["prosody"],"title":"Inference"},"AssistantMessage":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"from_text":{"type":"boolean","description":"Indicates if this message was inserted into the conversation as text from an [Assistant Input message](/reference/empathic-voice-interface-evi/chat/chat#send.Assistant%20Input.text)."},"id":{"type":"string","description":"ID of the assistant message. Allows the Assistant Message to be tracked and referenced."},"is_quick_response":{"type":"boolean","description":"Indicates if this message is a quick response or not."},"language":{"type":["string","null"],"default":null,"description":"Detected language of the message text."},"message":{"$ref":"#/components/schemas/ChatMessage","description":"Transcript of the message."},"models":{"$ref":"#/components/schemas/Inference","description":"Inference model results."},"type":{"type":"string","enum":["assistant_message"],"description":"The type of message sent through the socket; for an Assistant Message, this must be `assistant_message`.\n\nThis message contains both a transcript of the assistant's response and the expression measurement predictions of the assistant's audio output."}},"required":["from_text","is_quick_response","message","models","type"],"description":"**Transcript of the assistant's message.** Contains the message role, content, and optionally tool call information including the tool name, parameters, response requirement status, tool call ID, and tool type.","title":"AssistantMessage"},"AssistantProsody":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"id":{"type":"string","description":"Unique identifier for the segment."},"models":{"$ref":"#/components/schemas/Inference","description":"Inference model results."},"type":{"type":"string","enum":["assistant_prosody"],"description":"The type of message sent through the socket; for an Assistant Prosody message, this must be `assistant_PROSODY`.\n\nThis message the expression measurement predictions of the assistant's audio output."}},"required":["models","type"],"description":"**Expression measurement predictions of the assistant's audio output.** Contains inference model results including prosody scores for 48 emotions within the detected expression of the assistant's audio sample.","title":"AssistantProsody"},"AudioOutput":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"data":{"type":"string","format":"base64","description":"Base64 encoded audio output. This encoded audio is transmitted to the client, where it can be decoded and played back as part of the user interaction."},"id":{"type":"string","description":"ID of the audio output. Allows the Audio Output message to be tracked and referenced."},"index":{"type":"integer","description":"Index of the chunk of audio relative to the whole audio segment."},"type":{"type":"string","enum":["audio_output"]}},"required":["data","id","index","type"],"description":"**Base64 encoded audio output.** This encoded audio is transmitted to the client, where it can be decoded and played back as part of the user interaction. The returned audio format is WAV and the sample rate is 48kHz.\n\nContains the audio data, an ID to track and reference the audio output, and an index indicating the chunk position relative to the whole audio segment. See our [Audio Guide](/docs/speech-to-speech-evi/guides/audio) for more details on preparing and processing audio.","title":"AudioOutput"},"ChatMetadata":{"type":"object","properties":{"chat_group_id":{"type":"string","description":"ID of the Chat Group.\n\nUsed to resume a Chat when passed in the [resumed_chat_group_id](/reference/empathic-voice-interface-evi/chat/chat#request.query.resumed_chat_group_id) query parameter of a subsequent connection request. This allows EVI to continue the conversation from where it left off within the Chat Group.\n\nLearn more about [supporting chat resumability](/docs/empathic-voice-interface-evi/faq#does-evi-support-chat-resumability) from the EVI FAQ."},"chat_id":{"type":"string","description":"ID of the Chat session. Allows the Chat session to be tracked and referenced."},"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"request_id":{"type":["string","null"],"description":"ID of the initiating request."},"type":{"type":"string","enum":["chat_metadata"],"description":"The type of message sent through the socket; for a Chat Metadata message, this must be `chat_metadata`.\n\nThe Chat Metadata message is the first message you receive after establishing a connection with EVI and contains important identifiers for the current Chat session."}},"required":["chat_group_id","chat_id","request_id","type"],"description":"**The first message received after establishing a connection with EVI**, containing important identifiers for the current Chat session.\n\nIncludes the Chat ID (which allows the Chat session to be tracked and referenced) and the Chat Group ID (used to resume a Chat when passed in the `resumed_chat_group_id` query parameter of a subsequent connection request, allowing EVI to continue the conversation from where it left off within the Chat Group).","title":"ChatMetadata"},"Error":{"type":"object","properties":{"code":{"type":"string","description":"Error code. Identifies the type of error encountered."},"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"message":{"type":"string","description":"Detailed description of the error."},"request_id":{"type":["string","null"],"default":null,"description":"ID of the initiating request."},"slug":{"type":"string","description":"Short, human-readable identifier and description for the error. See a complete list of error slugs on the [Errors page](/docs/resources/errors)."},"type":{"type":"string","enum":["error"],"description":"The type of message sent through the socket; for a Web Socket Error message, this must be `error`.\n\nThis message indicates a disruption in the WebSocket connection, such as an unexpected disconnection, protocol error, or data transmission issue."}},"required":["code","message","slug","type"],"description":"**Indicates a disruption in the WebSocket connection**, such as an unexpected disconnection, protocol error, or data transmission issue.\n\nContains an error code identifying the type of error encountered, a detailed description of the error, and a short, human-readable identifier and description (slug) for the error.","title":"Error"},"UserInterruption":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"time":{"type":"integer","description":"Unix timestamp of the detected user interruption."},"type":{"type":"string","enum":["user_interruption"],"description":"The type of message sent through the socket; for a User Interruption message, this must be `user_interruption`.\n\nThis message indicates the user has interrupted the assistant's response. EVI detects the interruption in real-time and sends this message to signal the interruption event. This message allows the system to stop the current audio playback, clear the audio queue, and prepare to handle new user input."}},"required":["time","type"],"description":"**Indicates the user has interrupted the assistant's response.** EVI detects the interruption in real-time and sends this message to signal the interruption event.\n\nThis message allows the system to stop the current audio playback, clear the audio queue, and prepare to handle new user input. Contains a Unix timestamp of when the user interruption was detected. For more details, see our [Interruptibility Guide](/docs/speech-to-speech-evi/features/interruptibility)","title":"UserInterruption"},"MillisecondInterval":{"type":"object","properties":{"begin":{"type":"integer","description":"Start time of the interval in milliseconds."},"end":{"type":"integer","description":"End time of the interval in milliseconds."}},"required":["begin","end"],"title":"MillisecondInterval"},"UserMessage":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"from_text":{"type":"boolean","description":"Indicates if this message was inserted into the conversation as text from a [User Input](/reference/empathic-voice-interface-evi/chat/chat#send.User%20Input.text) message."},"interim":{"type":"boolean","description":"Indicates if this message contains an immediate and unfinalized transcript of the user's audio input. If it does, words may be repeated across successive UserMessage messages as our transcription model becomes more confident about what was said with additional context. Interim messages are useful to detect if the user is interrupting during audio playback on the client. Even without a finalized transcription, along with `UserInterrupt` messages, interim `UserMessages` are useful for detecting if the user is interrupting during audio playback on the client, signaling to stop playback in your application."},"language":{"type":["string","null"],"default":null,"description":"Detected language of the message text."},"message":{"$ref":"#/components/schemas/ChatMessage","description":"Transcript of the message."},"models":{"$ref":"#/components/schemas/Inference","description":"Inference model results."},"time":{"$ref":"#/components/schemas/MillisecondInterval","description":"Start and End time of user message."},"type":{"type":"string","enum":["user_message"]}},"required":["from_text","interim","message","models","time","type"],"description":"**Transcript of the user's message.** Contains the message role and content, along with a `from_text` field indicating if this message was inserted into the conversation as text from a `UserInput` message.\n\nIncludes an `interim` field indicating whether the transcript is provisional (words may be repeated or refined in subsequent `UserMessage` responses as additional audio is processed) or final and complete. Interim transcripts are only sent when the `verbose_transcription` query parameter is set to true in the initial handshake.","title":"UserMessage"},"AudioConfiguration":{"type":"object","properties":{"channels":{"type":"integer","description":"Number of audio channels."},"codec":{"type":["string","null"],"description":"Optional codec information."},"encoding":{"$ref":"#/components/schemas/Encoding","description":"Encoding format of the audio input, such as `linear16`."},"sample_rate":{"type":"integer","description":"Audio sample rate. Number of samples per second in the audio input, measured in Hertz."}},"required":["channels","encoding","sample_rate"],"title":"AudioConfiguration"},"BuiltInTool":{"type":"string","enum":["web_search","hang_up"],"title":"BuiltInTool"},"BuiltinToolConfig":{"type":"object","properties":{"fallback_content":{"type":["string","null"],"description":"Optional text passed to the supplemental LLM if the tool call fails. The LLM then uses this text to generate a response back to the user, ensuring continuity in the conversation."},"name":{"$ref":"#/components/schemas/BuiltInTool"}},"required":["name"],"title":"BuiltinToolConfig"},"Context":{"type":"object","properties":{"text":{"type":"string","description":"The context to be injected into the conversation. Helps inform the LLM's response by providing relevant information about the ongoing conversation.\n\nThis text will be appended to the end of user messages based on the chosen persistence level. For example, if you want to remind EVI of its role as a helpful weather assistant, the context you insert will be appended to the end of user messages as `{Context: You are a helpful weather assistant}`."},"type":{"$ref":"#/components/schemas/ContextType","description":"The persistence level of the injected context. Specifies how long the injected context will remain active in the session.\n\nThere are three possible context types:\n\n- **Persistent**: The context is appended to all user messages for the duration of the session.\n\n- **Temporary**: The context is appended only to the next user message.\n\n - **Editable**: The original context is updated to reflect the new context.\n\n If the type is not specified, it will default to `temporary`."}},"required":["text"],"title":"Context"},"Tool":{"type":"object","properties":{"description":{"type":["string","null"],"description":"An optional description of what the tool does, used by the supplemental LLM to choose when and how to call the function."},"fallback_content":{"type":["string","null"],"description":"Optional text passed to the supplemental LLM if the tool call fails. The LLM then uses this text to generate a response back to the user, ensuring continuity in the conversation."},"name":{"type":"string","description":"Name of the user-defined tool to be enabled."},"parameters":{"type":"string","description":"Parameters of the tool. Is a stringified JSON schema.\n\nThese parameters define the inputs needed for the tool's execution, including the expected data type and description for each input field. Structured as a JSON schema, this format ensures the tool receives data in the expected format."},"type":{"$ref":"#/components/schemas/ToolType","description":"Type of tool. Set to `function` for user-defined tools."}},"required":["name","parameters","type"],"title":"Tool"},"SessionSettings":{"type":"object","properties":{"audio":{"oneOf":[{"$ref":"#/components/schemas/AudioConfiguration"},{"type":"null"}],"default":null,"description":"Configuration details for the audio input used during the session. Ensures the audio is being correctly set up for processing.\n\nThis optional field is only required when the audio input is encoded in PCM Linear 16 (16-bit, little-endian, signed PCM WAV data). For detailed instructions on how to configure session settings for PCM Linear 16 audio, please refer to the [Session Settings section](/docs/empathic-voice-interface-evi/configuration#session-settings) on the EVI Configuration page."},"builtin_tools":{"type":["array","null"],"items":{"$ref":"#/components/schemas/BuiltinToolConfig"},"default":null,"description":"List of built-in tools to enable for the session.\n\nTools are resources used by EVI to perform various tasks, such as searching the web or calling external APIs. Built-in tools, like web search, are natively integrated, while user-defined tools are created and invoked by the user. To learn more, see our [Tool Use Guide](/docs/empathic-voice-interface-evi/tool-use).\n\nCurrently, the only built-in tool Hume provides is **Web Search**. When enabled, Web Search equips EVI with the ability to search the web for up-to-date information."},"context":{"oneOf":[{"$ref":"#/components/schemas/Context"},{"type":"null"}],"default":null,"description":"Field for injecting additional context into the conversation, which is appended to the end of user messages for the session.\n\nWhen included in a Session Settings message, the provided context can be used to remind the LLM of its role in every user message, prevent it from forgetting important details, or add new relevant information to the conversation.\n\nSet to `null` to clear injected context."},"custom_session_id":{"type":["string","null"],"default":null,"description":"Unique identifier for the session. Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions.\n\nIf included, the response sent from Hume to your backend will include this ID. This allows you to correlate frontend users with their incoming messages.\n\nIt is recommended to pass a `custom_session_id` if you are using a Custom Language Model. Please see our guide to [using a custom language model](/docs/empathic-voice-interface-evi/custom-language-model) with EVI to learn more."},"language_model_api_key":{"type":["string","null"],"default":null,"description":"Third party API key for the supplemental language model.\n\nWhen provided, EVI will use this key instead of Hume's API key for the supplemental LLM. This allows you to bypass rate limits and utilize your own API key as needed."},"metadata":{"type":["object","null"],"additionalProperties":{"description":"Any type"},"default":null},"system_prompt":{"type":["string","null"],"default":null,"description":"Instructions used to shape EVI's behavior, responses, and style for the session.\n\nWhen included in a Session Settings message, the provided Prompt overrides the existing one specified in the EVI configuration. If no Prompt was defined in the configuration, this Prompt will be the one used for the session.\n\nYou can use the Prompt to define a specific goal or role for EVI, specifying how it should act or what it should focus on during the conversation. For example, EVI can be instructed to act as a customer support representative, a fitness coach, or a travel advisor, each with its own set of behaviors and response styles.\n\nFor help writing a system prompt, see our [Prompting Guide](/docs/empathic-voice-interface-evi/prompting)."},"tools":{"type":["array","null"],"items":{"$ref":"#/components/schemas/Tool"},"default":null,"description":"List of user-defined tools to enable for the session.\n\nTools are resources used by EVI to perform various tasks, such as searching the web or calling external APIs. Built-in tools, like web search, are natively integrated, while user-defined tools are created and invoked by the user. To learn more, see our [Tool Use Guide](/docs/empathic-voice-interface-evi/tool-use)."},"type":{"type":"string","enum":["session_settings"],"description":"The type of message sent through the socket; must be `session_settings` for our server to correctly identify and process it as a Session Settings message.\n\nSession settings are temporary and apply only to the current Chat session. These settings can be adjusted dynamically based on the requirements of each session to ensure optimal performance and user experience.\n\nFor more information, please refer to the [Session Settings section](/docs/empathic-voice-interface-evi/configuration#session-settings) on the EVI Configuration page."},"variables":{"type":["object","null"],"additionalProperties":{"description":"Any type"},"default":null,"description":"This field allows you to assign values to dynamic variables referenced in your system prompt.\n\nEach key represents the variable name, and the corresponding value is the specific content you wish to assign to that variable within the session. While the values for variables can be strings, numbers, or booleans, the value will ultimately be converted to a string when injected into your system prompt.\n\nUsing this field, you can personalize responses based on session-specific details. For more guidance, see our [guide on using dynamic variables](/docs/speech-to-speech-evi/features/dynamic-variables)."},"voice_id":{"type":["string","null"],"default":null,"description":"Allows you to change the voice during an active chat. Updating the voice does not affect chat context or conversation history."}},"required":["type"],"description":"**Settings for this chat session.** Session settings are temporary and apply only to the current Chat session.\n\nThese settings can be adjusted dynamically based on the requirements of each session to ensure optimal performance and user experience. See our [Session Settings Guide](/docs/speech-to-speech-evi/configuration/session-settings) for a complete list of configurable settings.","title":"SessionSettings"},"SubscribeEvent":{"oneOf":[{"$ref":"#/components/schemas/AssistantEnd"},{"$ref":"#/components/schemas/AssistantMessage"},{"$ref":"#/components/schemas/AssistantProsody"},{"$ref":"#/components/schemas/AudioOutput"},{"$ref":"#/components/schemas/ChatMetadata"},{"$ref":"#/components/schemas/Error"},{"$ref":"#/components/schemas/UserInterruption"},{"$ref":"#/components/schemas/UserMessage"},{"$ref":"#/components/schemas/ToolCallMessage"},{"$ref":"#/components/schemas/ToolResponseMessage"},{"$ref":"#/components/schemas/ToolErrorMessage"},{"$ref":"#/components/schemas/SessionSettings"}],"title":"SubscribeEvent"},"ChatSubscribe":{"oneOf":[{"$ref":"#/components/schemas/SubscribeEvent"}],"title":"ChatSubscribe"},"AudioInput":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"data":{"type":"string","format":"base64","description":"Base64 encoded audio input to insert into the conversation.\n\nThe content of an Audio Input message is treated as the user's speech to EVI and must be streamed continuously. Pre-recorded audio files are not supported.\n\nFor optimal transcription quality, the audio data should be transmitted in small chunks.\n\nHume recommends streaming audio with a buffer window of 20 milliseconds (ms), or 100 milliseconds (ms) for web applications."},"type":{"type":"string","enum":["audio_input"],"description":"The type of message sent through the socket; must be `audio_input` for our server to correctly identify and process it as an Audio Input message.\n\nThis message is used for sending audio input data to EVI for processing and expression measurement. Audio data should be sent as a continuous stream, encoded in Base64."}},"required":["data","type"],"description":"**Base64 encoded audio input to insert into the conversation.** The content is treated as the user's speech to EVI and must be streamed continuously. Pre-recorded audio files are not supported. \n\nFor optimal transcription quality, the audio data should be transmitted in small chunks. Hume recommends streaming audio with a buffer window of `20` milliseconds (ms), or `100` milliseconds (ms) for web applications. See our [Audio Guide](/docs/speech-to-speech-evi/guides/audio) for more details on preparing and processing audio.","title":"AudioInput"},"UserInput":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"text":{"type":"string","description":"User text to insert into the conversation. Text sent through a User Input message is treated as the user's speech to EVI. EVI processes this input and provides a corresponding response.\n\nExpression measurement results are not available for User Input messages, as the prosody model relies on audio input and cannot process text alone."},"type":{"type":"string","enum":["user_input"],"description":"The type of message sent through the socket; must be `user_input` for our server to correctly identify and process it as a User Input message."}},"required":["text","type"],"description":"**User text to insert into the conversation.** Text sent through a User Input message is treated as the user's speech to EVI. EVI processes this input and provides a corresponding response.\n\nExpression measurement results are not available for User Input messages, as the prosody model relies on audio input and cannot process text alone.","title":"UserInput"},"AssistantInput":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null},"text":{"type":"string","description":"Assistant text to synthesize into spoken audio and insert into the conversation.\n\nEVI uses this text to generate spoken audio using our proprietary expressive text-to-speech model. Our model adds appropriate emotional inflections and tones to the text based on the user's expressions and the context of the conversation. The synthesized audio is streamed back to the user as an [Assistant Message](/reference/empathic-voice-interface-evi/chat/chat#receive.Assistant%20Message.type)."},"type":{"type":"string","enum":["assistant_input"],"description":"The type of message sent through the socket; must be `assistant_input` for our server to correctly identify and process it as an Assistant Input message."}},"required":["text","type"],"description":"**Assistant text to synthesize into spoken audio and insert into the conversation.** EVI uses this text to generate spoken audio using our proprietary expressive text-to-speech model.\n\nOur model adds appropriate emotional inflections and tones to the text based on the user's expressions and the context of the conversation. The synthesized audio is streamed back to the user as an Assistant Message.","title":"AssistantInput"},"PauseAssistantMessage":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"type":{"type":"string","enum":["pause_assistant_message"],"description":"The type of message sent through the socket; must be `pause_assistant_message` for our server to correctly identify and process it as a Pause Assistant message.\n\nOnce this message is sent, EVI will not respond until a [Resume Assistant message](/reference/empathic-voice-interface-evi/chat/chat#send.Resume%20Assistant%20Message.type) is sent. When paused, EVI won't respond, but transcriptions of your audio inputs will still be recorded."}},"required":["type"],"description":"**Pause responses from EVI.** Chat history is still saved and sent after resuming. Once this message is sent, EVI will not respond until a Resume Assistant message is sent.\n\nWhen paused, EVI won't respond, but transcriptions of your audio inputs will still be recorded. See our [Pause Response Guide](/docs/speech-to-speech-evi/features/pause-responses) for further details.","title":"PauseAssistantMessage"},"ResumeAssistantMessage":{"type":"object","properties":{"custom_session_id":{"type":["string","null"],"default":null,"description":"Used to manage conversational state, correlate frontend and backend data, and persist conversations across EVI sessions."},"type":{"type":"string","enum":["resume_assistant_message"],"description":"The type of message sent through the socket; must be `resume_assistant_message` for our server to correctly identify and process it as a Resume Assistant message.\n\nUpon resuming, if any audio input was sent during the pause, EVI will retain context from all messages sent but only respond to the last user message. (e.g., If you ask EVI two questions while paused and then send a `resume_assistant_message`, EVI will respond to the second question and have added the first question to its conversation context.)"}},"required":["type"],"description":"**Resume responses from EVI.** Chat history sent while paused will now be sent.\n\nUpon resuming, if any audio input was sent during the pause, EVI will retain context from all messages sent but only respond to the last user message. See our [Pause Response Guide](/docs/speech-to-speech-evi/features/pause-responses) for further details.","title":"ResumeAssistantMessage"},"ChatPublish":{"oneOf":[{"$ref":"#/components/schemas/AudioInput"},{"$ref":"#/components/schemas/SessionSettings"},{"$ref":"#/components/schemas/UserInput"},{"$ref":"#/components/schemas/AssistantInput"},{"$ref":"#/components/schemas/ToolResponseMessage"},{"$ref":"#/components/schemas/ToolErrorMessage"},{"$ref":"#/components/schemas/PauseAssistantMessage"},{"$ref":"#/components/schemas/ResumeAssistantMessage"}],"title":"ChatPublish"},"ChatChatIdConnectSubscribe":{"oneOf":[{"$ref":"#/components/schemas/SubscribeEvent"}],"title":"ChatChatIdConnectSubscribe"},"ControlPlanePublishEvent":{"oneOf":[{"$ref":"#/components/schemas/SessionSettings"},{"$ref":"#/components/schemas/UserInput"},{"$ref":"#/components/schemas/AssistantInput"},{"$ref":"#/components/schemas/ToolResponseMessage"},{"$ref":"#/components/schemas/ToolErrorMessage"},{"$ref":"#/components/schemas/PauseAssistantMessage"},{"$ref":"#/components/schemas/ResumeAssistantMessage"}],"title":"ControlPlanePublishEvent"},"ChatChatIdConnectPublish":{"oneOf":[{"$ref":"#/components/schemas/ControlPlanePublishEvent"}],"title":"ChatChatIdConnectPublish"}}}}