asyncapi: 3.0.0

# Trigger redeploy
info:
  title: AssemblyAI Voice Agent API
  version: 1.0.0
  description: |
    Real-time voice conversation API over a single WebSocket. The agent listens to the user,
    transcribes, reasons, and speaks a response back — all in one connection. Clients stream
    PCM16 audio and receive audio, transcripts, and tool calls in return.

servers:
  production:
    host: agents.assemblyai.com
    protocol: wss
    description: Production Voice Agent WebSocket endpoint

channels:
  voiceAgent:
    address: /v1/ws
    description: |
      
      Connect to the Voice Agent API to run a real-time voice conversation. The client streams
      PCM16 audio to the server and receives the agent's spoken response (also PCM16), along with
      transcripts, tool calls, and lifecycle events.

      After the WebSocket opens, send a [`session.update`](#sendSessionUpdate) as your
      first message. You have two ways to configure the agent:

      - **Stored agent.** Send `{ "agent_id": "<id>" }` as the only field in `session`
        to bind to a reusable agent created via the
        [Agents REST API](https://www.assemblyai.com/docs/api-reference/voice-agent-api/create-agent). The stored
        `system_prompt`, `greeting`, `tools`, `input`, and `output` are applied
        server-side.
      - **Inline configuration.** Omit `agent_id` and send `system_prompt`, `greeting`,
        `tools`, `input`, and `output` directly. Useful for one-off or fully dynamic
        agents.

      The two modes are mutually exclusive. See
      [Deploy your agent](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/deploy) and
      [Inline session configuration](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/session-configuration)
      for details, or jump to the [Voice Agent API overview](https://www.assemblyai.com/docs/voice-agents/voice-agent-api)
      for the full event flow and a runnable quickstart.
    servers:
      - $ref: "#/servers/production"
    parameters:
      ApiKey:
        description: >-
          Pass your API key as a Bearer token in the `Authorization` header on the
          WebSocket upgrade request. For browser apps (which can't set custom headers
          on WebSockets), generate a
          [temporary token](https://www.assemblyai.com/docs/api-reference/voice-agent-api/generate-voice-agent-token)
          and pass it via the `token` query parameter instead. See
          [Browser integration](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/browser-integration).
        location: $message.header#/Authorization
        examples:
          - Bearer YOUR_ASSEMBLYAI_API_KEY

      token:
        description: >-
          Temporary authentication token for client-side connections. Generate one with
          [`GET /v1/token`](https://www.assemblyai.com/docs/api-reference/voice-agent-api/generate-voice-agent-token)
          on your server and pass it here so you don't expose your permanent API key in
          the browser. Each token is one-time use.
        location: $message.payload#/token

    messages:
      sessionUpdate:
        $ref: "#/components/messages/SessionUpdate"
      sessionResume:
        $ref: "#/components/messages/SessionResume"
      sessionEnd:
        $ref: "#/components/messages/SessionEnd"
      inputAudio:
        $ref: "#/components/messages/InputAudio"
      toolResult:
        $ref: "#/components/messages/ToolResult"
      replyCreate:
        $ref: "#/components/messages/ReplyCreate"

      sessionReady:
        $ref: "#/components/messages/SessionReady"
      sessionUpdated:
        $ref: "#/components/messages/SessionUpdated"
      sessionEnded:
        $ref: "#/components/messages/SessionEnded"
      sessionError:
        $ref: "#/components/messages/SessionError"
      inputSpeechStarted:
        $ref: "#/components/messages/InputSpeechStarted"
      inputSpeechStopped:
        $ref: "#/components/messages/InputSpeechStopped"
      transcriptUserDelta:
        $ref: "#/components/messages/TranscriptUserDelta"
      transcriptUser:
        $ref: "#/components/messages/TranscriptUser"
      replyStarted:
        $ref: "#/components/messages/ReplyStarted"
      replyAudio:
        $ref: "#/components/messages/ReplyAudio"
      transcriptAgent:
        $ref: "#/components/messages/TranscriptAgent"
      replyDone:
        $ref: "#/components/messages/ReplyDone"
      toolCall:
        $ref: "#/components/messages/ToolCall"

operations:
  sendSessionUpdate:
    action: send
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionUpdate"
    description: |
      Configure the session. Send immediately on connect — before `session.ready` — to either:

      - **Bind to a stored agent** by sending `{ "agent_id": "<id>" }` as the only field in
        `session`. The agent's stored `system_prompt`, `greeting`, `tools`, `input`, and
        `output` are loaded server-side. Create stored agents with
        [`POST /v1/agents`](https://www.assemblyai.com/docs/api-reference/voice-agent-api/create-agent).
      - **Configure inline** by sending any combination of `system_prompt`, `greeting`,
        `tools`, `input`, and `output` (omit `agent_id`).

      `agent_id` is mutually exclusive with the inline fields; sending both raises a
      validation error.

      Can also be sent mid-conversation to update mutable fields (e.g. `system_prompt`,
      `input.turn_detection`). `greeting` and `output` are immutable after `session.ready`
      and changing them returns `immutable_field`.

  sendSessionResume:
    action: send
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionResume"
    description: |
      Resume a previous session using the `session_id` from a prior `session.ready`. Preserves
      conversation context across dropped connections. Sessions are held for 30 seconds after
      every disconnection.

  sendSessionEnd:
    action: send
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionEnd"
    description: |
      Cleanly end the session. The server emits a final `session.ended` and closes the
      WebSocket; the `session_id` is dead immediately and cannot be resumed. Use this
      instead of just closing the socket when the call is over — closing the socket without
      sending `session.end` leaves the session resumable (and billable) for 30 seconds.

  sendInputAudio:
    action: send
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/inputAudio"
    description: |
      Stream a chunk of user audio to the agent. Only send `input.audio` after `session.ready`.
      See [Audio format](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/audio-format) for the expected
      encoding (base64-encoded, mono). Supports `audio/pcm` (24 kHz), `audio/pcmu` (8 kHz), and `audio/pcma` (8 kHz).

  sendToolResult:
    action: send
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/toolResult"
    description: |
      Return a tool result to the agent. Send this inside your `reply.done` handler — not
      immediately on `tool.call`. See
      [Tool calling](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/tools/overview).

  sendReplyCreate:
    action: send
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/replyCreate"
    description: |
      Ask the agent to generate a reply right now, optionally with one-shot
      `instructions`. Primary use case: deliver status updates while a `hold`-mode
      tool call is in flight. See
      [Tool calling — hold mode](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/tools/overview#hold).

  receiveSessionReady:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionReady"
    description: Session is established. Save `session_id` for reconnection and start streaming audio.

  receiveSessionUpdated:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionUpdated"
    description: Sent after a `session.update` is applied successfully.

  receiveSessionEnded:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionEnded"
    description: |
      Final event emitted on every clean teardown, right before the WebSocket closes.
      Sent when the client sends `session.end`, when the session hits
      `max_session_duration_seconds`, when the server hits an unrecoverable error, or
      when the 30-second grace window after a disconnect expires.

  receiveSessionError:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/sessionError"
    description: A session- or protocol-level error occurred.

  receiveInputSpeechStarted:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/inputSpeechStarted"
    description: Turn detection determined the user has started speaking.

  receiveInputSpeechStopped:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/inputSpeechStopped"
    description: Turn detection determined the user has stopped speaking.

  receiveTranscriptUserDelta:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/transcriptUserDelta"
    description: Partial transcript of the user's utterance, updating in real-time.

  receiveTranscriptUser:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/transcriptUser"
    description: Final transcript of the user's utterance.

  receiveReplyStarted:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/replyStarted"
    description: Agent has begun generating a response.

  receiveReplyAudio:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/replyAudio"
    description: |
      A chunk of the agent's spoken response (base64 PCM16). Decode and play immediately.
      See [Audio format](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/audio-format#playing-output-audio)
      for playback guidance.

  receiveTranscriptAgent:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/transcriptAgent"
    description: Full text of the agent's response, delivered after all audio for the reply has been sent.

  receiveReplyDone:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/replyDone"
    description: Agent has finished speaking. Send any accumulated `tool.result` events here.

  receiveToolCall:
    action: receive
    channel:
      $ref: "#/channels/voiceAgent"
    messages:
      - $ref: "#/channels/voiceAgent/messages/toolCall"
    description: |
      Agent wants to invoke a registered tool. Execute the tool, then send the result with
      `tool.result` after `reply.done` fires.

components:
  messages:
    SessionUpdate:
      name: SessionUpdate
      title: Update Session
      summary: |
        Client message to configure the session. Either bind to a stored agent by sending
        `agent_id`, or configure an inline agent with `system_prompt`, `greeting`, `input`,
        `output`, and `tools`. The two modes are mutually exclusive.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionUpdatePayload"
      examples:
        - name: BindStoredAgent
          summary: Bind the session to a stored agent by `agent_id`.
          payload:
            type: session.update
            session:
              agent_id: 7ad24396-b822-4dca-871a-be9cc4781cf9
        - name: InlineConfiguration
          summary: Configure the agent inline at connect time.
          payload:
            type: session.update
            session:
              system_prompt: You are a concise assistant.
              greeting: Hi — how can I help?
              input:
                format:
                  encoding: audio/pcm
                turn_detection:
                  vad_threshold: 0.5
              output:
                voice: anna
                format:
                  encoding: audio/pcm
                volume: 100
              tools:
                - type: function
                  name: get_weather
                  description: Get weather for a city
                  parameters:
                    type: object
                    properties:
                      city:
                        type: string
                    required:
                      - city

    SessionResume:
      name: SessionResume
      title: Resume Session
      summary: Client message to resume a previous session by `session_id`.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionResumePayload"
      examples:
        - payload:
            type: session.resume
            session_id: sess_abc123

    SessionEnd:
      name: SessionEnd
      title: End Session
      summary: |
        Client message to cleanly end the session. The server emits a final `session.ended`
        and closes the WebSocket; the `session_id` is dead immediately and cannot be resumed.
        Use this instead of just closing the socket to stop billing right away — closing the
        socket without `session.end` leaves the session resumable (and billable) for 30 seconds.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionEndPayload"
      examples:
        - payload:
            type: session.end

    InputAudio:
      name: InputAudio
      title: Input Audio Chunk
      summary: Client streams a chunk of PCM16 audio as base64.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/InputAudioPayload"
      examples:
        - payload:
            type: input.audio
            audio: EAAgADAAQAAwACAAEAAAAPD/4P/Q/8D/

    ToolResult:
      name: ToolResult
      title: Tool Result
      summary: Client returns the result of a tool invocation to the agent.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/ToolResultPayload"
      examples:
        - payload:
            type: tool.result
            call_id: call_abc123
            result: '{"temp_c": 22, "description": "Sunny"}'

    ReplyCreate:
      name: ReplyCreate
      title: Reply Create
      summary: Client asks the agent to generate a reply now, optionally with one-shot instructions.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/ReplyCreatePayload"
      examples:
        - payload:
            type: reply.create
            instructions: Let the customer know we're still processing the transfer.

    SessionReady:
      name: SessionReady
      title: Session Ready
      summary: Server confirms the session is established and ready for audio.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionReadyPayload"
      examples:
        - payload:
            type: session.ready
            session_id: sess_abc123

    SessionUpdated:
      name: SessionUpdated
      title: Session Updated
      summary: Server acknowledges that a `session.update` was applied successfully.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionUpdatedPayload"
      examples:
        - payload:
            type: session.updated

    SessionEnded:
      name: SessionEnded
      title: Session Ended
      summary: |
        Final event emitted on every clean teardown, right before the WebSocket closes. Sent
        when the client sends `session.end`, the session hits `max_session_duration_seconds`,
        the server hits an unrecoverable error, or the 30-second grace window after a
        disconnect expires.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionEndedPayload"
      examples:
        - payload:
            type: session.ended
            session_duration_seconds: 42.7
            audio_duration_seconds: 38.2
            timestamp: 1717180000.123

    SessionError:
      name: SessionError
      title: Session Error
      summary: Server reports a session- or protocol-level error.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/SessionErrorPayload"
      examples:
        - payload:
            type: session.error
            code: invalid_format
            message: Invalid message format

    InputSpeechStarted:
      name: InputSpeechStarted
      title: User Started Speaking
      summary: Server signals that turn detection determined the user started speaking.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/InputSpeechStartedPayload"
      examples:
        - payload:
            type: input.speech.started

    InputSpeechStopped:
      name: InputSpeechStopped
      title: User Stopped Speaking
      summary: Server signals that turn detection determined the user stopped speaking.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/InputSpeechStoppedPayload"
      examples:
        - payload:
            type: input.speech.stopped

    TranscriptUserDelta:
      name: TranscriptUserDelta
      title: User Transcript Delta
      summary: Partial transcript of the user's current utterance.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/TranscriptUserDeltaPayload"
      examples:
        - payload:
            type: transcript.user.delta
            text: What's the weather in

    TranscriptUser:
      name: TranscriptUser
      title: User Transcript
      summary: Final transcript of the user's utterance.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/TranscriptUserPayload"
      examples:
        - payload:
            type: transcript.user
            text: What's the weather in Tokyo?
            item_id: item_abc123

    ReplyStarted:
      name: ReplyStarted
      title: Reply Started
      summary: Agent has begun generating a reply.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/ReplyStartedPayload"
      examples:
        - payload:
            type: reply.started
            reply_id: reply_abc123

    ReplyAudio:
      name: ReplyAudio
      title: Reply Audio Chunk
      summary: A chunk of the agent's spoken response as base64 PCM16.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/ReplyAudioPayload"
      examples:
        - payload:
            type: reply.audio
            data: EAAgADAAQAAwACAAEAAAAPD/4P/Q/8D/

    TranscriptAgent:
      name: TranscriptAgent
      title: Agent Transcript
      summary: Text of the agent's response, sent after all reply audio has been delivered.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/TranscriptAgentPayload"
      examples:
        - payload:
            type: transcript.agent
            text: It's currently 22°C and sunny in Tokyo.
            reply_id: reply_abc123
            item_id: item_abc123
            interrupted: false

    ReplyDone:
      name: ReplyDone
      title: Reply Done
      summary: |
        Agent has finished speaking. If the user barged in, `status` is `"interrupted"`. Send
        accumulated `tool.result` events on this event.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/ReplyDonePayload"
      examples:
        - payload:
            type: reply.done

    ToolCall:
      name: ToolCall
      title: Tool Call
      summary: Agent wants to invoke a registered tool.
      contentType: application/json
      payload:
        $ref: "#/components/schemas/ToolCallPayload"
      examples:
        - payload:
            type: tool.call
            call_id: call_abc123
            name: get_weather
            arguments:
              location: Tokyo

  schemas:
    SessionConfig:
      type: object
      description: |
        Session configuration fields. All fields are optional — only include the ones you want to change.

        There are two ways to configure a session:

        - **Bind to a stored agent.** Send `agent_id` (and only `agent_id`) in your first
          `session.update` to load a reusable agent created via the
          [Agents REST API](https://www.assemblyai.com/docs/api-reference/voice-agent-api/create-agent). The agent's
          stored `system_prompt`, `greeting`, `tools`, `input`, and `output` are
          applied server-side.
        - **Configure inline.** Omit `agent_id` and send any combination of
          `system_prompt`, `greeting`, `tools`, `input`, and `output` directly.

        `agent_id` is mutually exclusive with the inline fields — sending both in the
        same `session.update` is rejected. See
        [Deploy your agent](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/deploy) for details.
      required: []
      properties:
        agent_id:
          type: string
          description: |
            ID of a stored agent (from
            [`POST /v1/agents`](https://www.assemblyai.com/docs/api-reference/voice-agent-api/create-agent))
            to bind this session to. When set, the agent's stored configuration is
            applied and you must not send `system_prompt`, `greeting`, `tools`,
            `input`, or `output` in the same `session.update`. Can only be set in
            the first `session.update`, before `session.ready`.
          examples:
            - 7ad24396-b822-4dca-871a-be9cc4781cf9
        system_prompt:
          type: string
          description: The agent's personality and context. Can be updated mid-session. Mutually exclusive with `agent_id`.
        greeting:
          type: string
          description: |
            What the agent says at the start of the conversation. Sent directly to
            the TTS engine and spoken verbatim. It is NOT run through the LLM,
            so write the exact words you want the user to hear. Immutable after
            `session.ready`. Omit to have the agent wait silently for the user to
            speak first.
        input:
          $ref: "#/components/schemas/SessionInputConfig"
        output:
          $ref: "#/components/schemas/SessionOutputConfig"
        tools:
          type: array
          description: Tool definitions. See [Tool calling](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/tools/overview).
          items:
            $ref: "#/components/schemas/ToolDefinition"

    SessionInputConfig:
      type: object
      description: Configuration for the user audio input stream.
      required: []
      properties:
        format:
          $ref: "#/components/schemas/AudioFormat"
        keyterms:
          type: array
          items:
            type: string
          description: List of rare or domain-specific terms to boost in transcription (e.g. names, brands, jargon). See [Key terms](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/session-configuration#key-terms).
        turn_detection:
          $ref: "#/components/schemas/TurnDetection"

    SessionOutputConfig:
      type: object
      description: Configuration for the agent audio output stream.
      required: []
      properties:
        voice:
          type: string
          description: Voice used for the agent's speech. See [Voices](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/voices).
          default: anna
          examples:
            - anna
        format:
          $ref: "#/components/schemas/AudioFormat"
        volume:
          type: number
          minimum: 0
          maximum: 100
          description: Playback volume for the agent's speech. `0` is silent, `100` is loudest. If omitted, the voice plays at its native level. Mutable mid-session.
          examples:
            - 100

    AudioFormat:
      type: object
      description: Audio format configuration. The encoding determines the sample rate. See [Audio format](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/audio-format).
      required: []
      properties:
        encoding:
          type: string
          enum:
            - audio/pcm
            - audio/pcmu
            - audio/pcma
          default: audio/pcm
          description: "Audio encoding. `audio/pcm` (PCM16, 16-bit little-endian, 24 kHz), `audio/pcmu` (G.711 μ-law, 8 kHz), or `audio/pcma` (G.711 A-law, 8 kHz)."

    ToolDefinition:
      type: object
      description: A single tool the agent can call.
      properties:
        type:
          type: string
          const: function
          description: Always `function`.
        name:
          type: string
          description: Name of the tool, referenced by `tool.call`.
        description:
          type: string
          description: Human-readable description of what the tool does.
        parameters:
          type: object
          description: |
            JSON Schema describing the tool's arguments. Must be a JSON Schema object
            (`{"type": "object", "properties": {...}, "required": [...]}`).
            Passed to the agent's LLM verbatim. The server does NOT validate the schema
            at `session.update` time, so a malformed schema is accepted but produces
            unpredictable tool calls. Validate your schema before sending.
            Every property should have a `description`. That string is how the model
            extracts argument values from user speech.
          additionalProperties: true
          required: []
          properties: {}
        execution_mode:
          type: string
          enum:
            - interactive
            - hold
          default: interactive
          description: |
            How the agent waits for the result.
            `"interactive"` (default): the agent speaks a short transition phrase
            (e.g. "let me check that") while the tool runs, then delivers the result
            conversationally. Use for short tools (DB lookups, REST calls).
            `"hold"`: the agent stays silent while the tool runs and suppresses replies
            triggered by user speech. Send `reply.create` to deliver status updates
            during the hold. When `tool.result` arrives, the agent gives a brief,
            direct delivery of the result. Use for long-running tools (transfers,
            escalations, async jobs). See
            [Tool calling — execution modes](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/tools/overview#execution-modes).
        timeout_seconds:
          type: number
          format: float
          minimum: 1
          maximum: 300
          default: 120
          description: How long to wait for the matching `tool.result` before timing out.
      required:
        - type
        - name
        - description
        - parameters

    TurnDetection:
      type: object
      description: Configures turn detection sensitivity, end-of-turn detection, and barge-in behavior. Lives under `session.input.turn_detection`.
      required: []
      properties:
        vad_threshold:
          type: number
          format: float
          minimum: 0
          maximum: 1
          default: 0.5
          description: Speech detection sensitivity (0.0–1.0). Lower = more sensitive to speech.
        min_silence:
          type: integer
          minimum: 50
          maximum: 10000
          default: 1000
          description: Minimum silence to consider a confident end-of-turn, in milliseconds. Must be less than `max_silence`.
        max_silence:
          type: integer
          minimum: 50
          maximum: 10000
          default: 3000
          description: Maximum silence before forcing end-of-turn, in milliseconds. Must be greater than `min_silence`.
        interrupt_response:
          type: boolean
          default: true
          description: Whether user speech interrupts the agent. Set `false` to disable barge-in.

    SessionUpdatePayload:
      type: object
      properties:
        type:
          type: string
          const: session.update
        session:
          $ref: "#/components/schemas/SessionConfig"
      required:
        - type
        - session

    SessionResumePayload:
      type: object
      properties:
        type:
          type: string
          const: session.resume
        session_id:
          type: string
          description: The `session_id` from a previous `session.ready` event.
      required:
        - type
        - session_id

    SessionEndPayload:
      type: object
      properties:
        type:
          type: string
          const: session.end
      required:
        - type

    InputAudioPayload:
      type: object
      properties:
        type:
          type: string
          const: input.audio
        audio:
          type: string
          description: Base64-encoded audio chunk in the configured input encoding.
      required:
        - type
        - audio

    ToolResultPayload:
      type: object
      properties:
        type:
          type: string
          const: tool.result
        call_id:
          type: string
          description: The `call_id` from the `tool.call` event you are responding to.
        result:
          type: string
          description: |
            JSON-encoded string containing the tool result. Always a string,
            not a nested object. Use `json.dumps(...)` (Python) or `JSON.stringify(...)` (JS)
            on the payload before sending.
      required:
        - type
        - call_id
        - result

    ReplyCreatePayload:
      type: object
      properties:
        type:
          type: string
          const: reply.create
        instructions:
          type: string
          description: |
            Optional one-shot instructions the agent uses to compose this reply.
            Does not modify `system_prompt`. Useful for status updates during a
            `hold`-mode tool call. See
            [Tool calling — execution modes](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/tools/overview#execution-modes).
      required:
        - type

    SessionReadyPayload:
      type: object
      properties:
        type:
          type: string
          const: session.ready
        session_id:
          type: string
          description: Unique identifier for this session. Save this to reconnect with `session.resume`.
      required:
        - type
        - session_id

    SessionUpdatedPayload:
      type: object
      properties:
        type:
          type: string
          const: session.updated
      required:
        - type

    SessionEndedPayload:
      type: object
      properties:
        type:
          type: string
          const: session.ended
        session_duration_seconds:
          type: number
          description: Total wall-clock duration of the session, in seconds.
        audio_duration_seconds:
          type: number
          nullable: true
          description: Total audio streamed in by the client, in seconds. `null` if no audio was streamed.
        timestamp:
          type: number
          description: Unix epoch seconds when the server emitted the event.
      required:
        - type
        - session_duration_seconds
        - timestamp

    SessionErrorPayload:
      type: object
      properties:
        type:
          type: string
          enum:
            - session.error
            - error
          description: |
            `session.error` for session/protocol errors and `error` for connection-level errors.
        code:
          type: string
          enum:
            - UNAUTHORIZED
            - FORBIDDEN
            - INTERNAL_ERROR
            - server_error
            - session_not_found
            - session_forbidden
            - session_expired
            - agent_init_failed
            - agent_timeout
            - invalid_format
            - invalid_audio
            - invalid_value
            - immutable_field
            - invalid_config
          description: |
            Machine-readable error code. See [error codes](https://www.assemblyai.com/docs/voice-agents/voice-agent-api/events-reference#sessionerror)
            for the full table grouped by lifecycle stage.
        message:
          type: string
          description: Human-readable error description.
        timestamp:
          type: string
          format: date-time
          description: ISO-8601 timestamp of the error.
        param:
          type: string
          description: Name of the offending field when applicable (e.g. on `session.update` validation failures).
      required:
        - type
        - code
        - message

    InputSpeechStartedPayload:
      type: object
      properties:
        type:
          type: string
          const: input.speech.started
      required:
        - type

    InputSpeechStoppedPayload:
      type: object
      properties:
        type:
          type: string
          const: input.speech.stopped
      required:
        - type

    TranscriptUserDeltaPayload:
      type: object
      properties:
        type:
          type: string
          const: transcript.user.delta
        text:
          type: string
          description: Partial transcript of what the user is saying.
      required:
        - type
        - text

    TranscriptUserPayload:
      type: object
      properties:
        type:
          type: string
          const: transcript.user
        text:
          type: string
          description: Final transcript of the user's utterance.
        item_id:
          type: string
          description: Conversation item ID.
      required:
        - type
        - text
        - item_id

    ReplyStartedPayload:
      type: object
      properties:
        type:
          type: string
          const: reply.started
        reply_id:
          type: string
          description: ID of this reply.
      required:
        - type
        - reply_id

    ReplyAudioPayload:
      type: object
      properties:
        type:
          type: string
          const: reply.audio
        data:
          type: string
          description: Base64-encoded audio chunk in the configured output encoding.
      required:
        - type
        - data

    TranscriptAgentPayload:
      type: object
      properties:
        type:
          type: string
          const: transcript.agent
        text:
          type: string
          description: What the agent said. If interrupted, trimmed to the point of interruption.
        reply_id:
          type: string
          description: ID of the reply this transcript belongs to.
        item_id:
          type: string
          description: Conversation item ID.
        interrupted:
          type: boolean
          description: Whether the user interrupted the agent mid-response.
      required:
        - type
        - text
        - reply_id
        - item_id
        - interrupted

    ReplyDonePayload:
      type: object
      properties:
        type:
          type: string
          const: reply.done
        status:
          type: string
          enum:
            - completed
            - interrupted
          description: |
            `"completed"` for normal completion, `"interrupted"` if the user barged in.
      required:
        - type
        - status

    ToolCallPayload:
      type: object
      properties:
        type:
          type: string
          const: tool.call
        call_id:
          type: string
          description: Include this value in the corresponding `tool.result`.
        name:
          type: string
          description: Name of the tool the agent is invoking.
        arguments:
          type: object
          description: Arguments to pass to the tool, as a dictionary.
          additionalProperties: true
          required: []
          properties: {}
      required:
        - type
        - call_id
        - name
        - arguments
