asyncapi: 3.0.0
info:
  title: allmodels realtime (WebSocket) API
  version: 1.0.0
  description: |
    AllModels provides one API for text-to-speech and speech-to-text across
    multiple providers. This document describes the realtime WebSocket surface.
    See the HTTP API reference for handshake parameters and HTTP error responses.

    `receive` operations list frames sent by clients. `send` operations list
    frames returned by AllModels.

    ## Authentication
    AllModels `/v1` channels use `Authorization: Bearer <key>`.
    ElevenLabs-compatible `/el` channels additionally accept
    `xi-api-key: <key>`. OpenAI SDK clients use `Authorization: Bearer <key>`
    on `/oai/realtime`; browser clients may use an
    `openai-insecure-api-key.<key>` WebSocket subprotocol token. Realtime voice
    clients may instead use a short-lived `client_secret` or signed URL issued
    by your application server.

    Authentication failures close the WebSocket with code `4401` (invalid key)
    or `4403` (provider not allowed); the close reason contains
    `{ "error": "<code>" }`.

    ## Models and providers
    Use `{author}/{modelName}` model slugs where supported. Each channel lists
    the relevant model and provider examples. Use the provider-preference query params
    (`provider_order`, `provider_only`, `provider_ignore`, `allow_fallbacks`)
    to restrict or prioritize providers.
defaultContentType: application/json
servers:
  production:
    host: api.allmodels.io
    protocol: wss
    description: Production API.
  development:
    host: api.dev.allmodels.io
    protocol: wss
    description: Development API.
channels:
  nativeTts:
    address: /v1/tts
    title: Native streaming TTS
    summary: Stream text in, receive binary audio out, using a simple JSON command protocol.
    description: |
      Send JSON command frames (`text`, `flush`, and `close`) as WebSocket text
      frames. AllModels returns **binary** audio frames plus typed JSON text events for
      readiness, completion, errors, provider status, and policy checks.

      Handshake: HTTP GET with `Upgrade: websocket`. Authenticate with
      `Authorization: Bearer <key>`. Text frames sent while the session
      is starting are accepted and processed when it becomes ready.

      Provider-specific settings ride the query string: either directly
      (`sample_rate=24000`) or in bracket notation
      (`provider_options[sample_rate]=24000`, which wins when both forms set
      the same key). See the OpenAPI document for the per-provider option
      schemas.

      The WebSocket binding below cannot express text-vs-binary framing:
      client command and server event frames are JSON **text** frames; audio arrives as
      **binary** frames (raw 16-bit little-endian PCM, or MP3 when
      `?format=mp3` is set and the provider can emit it).

      Close-code semantics:
      - `1000` — normal completion, normalized to reason `tts_complete` after
        `allmodels.tts_session_ended` regardless of the selected backend's native
        completion signal or clean close code/reason.
      - `1003` — malformed client frame (reason `invalid_tts_frame`).
      - `1009` — a `text` frame exceeded 2000 characters
        (reason `tts_text_frame_too_large`).
      - `1011` — the provider session could not be established or failed while
        streaming. Fatal streaming errors use reason `upstream_error`; setup reason
        text is diagnostic and may change. Applications should branch on the close
        code rather than other reason text.
      - `4401` / `4403` — authentication / provider-allowlist failure.
    bindings:
      ws:
        method: GET
        query:
          type: object
          description: Handshake query parameters. Unreserved keys are treated as provider options; `provider_options[<name>]=<value>` bracket params take precedence over the bare form. Recognized enum and boolean option values are case-insensitive and are normalized to each provider's wire spelling; free-form values such as prompts, keyterms, and voice IDs retain their original case.
          required:
            - model
          properties:
            provider:
              type: string
              description: Pin a specific TTS provider (e.g. `elevenlabs`, `deepgram`, `grok`, `cartesia`).
            voice:
              type: string
              description: Provider voice id (e.g. `eve` for Grok, `EXAVITQu4vr4xnSDxMaL` for ElevenLabs). Required when the routed model's provider requires a voice — a missing value is then refused 400 `voice_required` before the upgrade (see `defaults.tts.voice` on GET /v1/providers for a recommended value); otherwise an omitted voice is forwarded as absent, never substituted.
            model:
              type: string
              description: Model id, bare (`eleven_flash_v2_5`) or `{author}/{modelName}` slug (`elevenlabs/eleven_flash_v2_5`).
            format:
              type: string
              enum:
                - mp3
                - pcm
              description: Output audio container for the binary frames.
            auto_flush:
              type: string
              default: "true"
              description: Auto-flush buffered text. Any value except the literal `false` enables it.
            provider_order:
              type: string
              description: Comma-separated or repeated provider ids in preferred order.
            provider_only:
              type: string
              description: Comma-separated or repeated provider ids; only these may serve the request.
            provider_ignore:
              type: string
              description: Comma-separated or repeated provider ids that must not serve the request.
            allow_fallbacks:
              type: string
              description: Set to `false` to prevent fallback to another provider.
          additionalProperties: true
        headers:
          type: object
          description: Handshake headers (besides the standard WebSocket upgrade headers).
          properties:
            authorization:
              type: string
              description: AllModels API key as `Bearer <key>`.
        bindingVersion: 0.1.0
    messages:
      nativeTtsText:
        $ref: "#/components/messages/nativeTtsText"
      nativeTtsFlush:
        $ref: "#/components/messages/nativeTtsFlush"
      nativeTtsClose:
        $ref: "#/components/messages/nativeTtsClose"
      sessionReady:
        $ref: "#/components/messages/sessionReady"
      nativeTtsAudioChunk:
        $ref: "#/components/messages/nativeTtsAudioChunk"
      nativeTtsSessionEnded:
        $ref: "#/components/messages/nativeTtsSessionEnded"
      nativeTtsError:
        $ref: "#/components/messages/nativeTtsError"
      nativeTtsProviderStatus:
        $ref: "#/components/messages/nativeTtsProviderStatus"
      nativeTtsPolicyCheck:
        $ref: "#/components/messages/nativeTtsPolicyCheck"
  nativeStt:
    address: /v1/stt
    title: Native streaming STT
    summary: Stream binary audio in and receive normalized, sequenced STT events.
    description: |
      The client streams **binary** audio frames and receives normalized
      `stt.*` JSON events with stable event/turn ids and gapless sequence
      numbers. Provider activity/turn decisions and router-VAD detections use
      a unified `stt.signal` object. `origin` distinguishes provider-decided
      from router-decided signals. Detection stays separate from segmentation:
      router VAD may emit `started`/`end` signals (`turn_scope=utterance`,
      `basis=[vad]`, frame-derived audio clocks), then emit a distinct
      `stt.audio.committed` acknowledgement. Transcript commits themselves
      never manufacture signals.

      Text controls are `stt.audio.commit`, `stt.audio.clear`, and
      `stt.session.close`. Set `event_format=provider` for provider-native
      passthrough in both directions. Provider mode forwards upstream server
      frames without event mapping, timestamp rewriting, deduplication, or
      router-generated JSON frames. A provider Blob retains its bytes but is
      represented as an ArrayBuffer because the Worker WebSocket API cannot
      send Blob values. Provider mode requires a direct upstream realtime
      WebSocket.

      Handshake: HTTP GET with `Upgrade: websocket`, authenticated with
      `Authorization: Bearer <key>`. `audio_format` is required and
      describes the unchanged raw bytes. Audio sent while the session is
      starting is accepted. The router never resamples or transcodes streaming
      audio. Unknown bare query parameters are rejected. Provider-native
      overrides use `provider_options[<name>]=<value>` and win over portable
      tuning, except that conflicting encoding/sample-rate/channel declarations
      fail with `conflicting_audio_format`.

      Portable defaults are `commit_strategy=auto` and `turn_detection=auto`.
      `interim_results` has no global default: when it is not sent it falls
      back to the selected model family's own behavior — `false` for Deepgram
      Flux, `true` for the other streaming families. Each family advertises
      only the portable options it can translate to a native parameter, and an
      explicit option outside that set fails before audio with
      `unsupported_stt_option`. Router VAD supplies acoustic automatic commits
      for OpenAI realtime Whisper and Cartesia `ink-whisper`; it inspects only
      declared little-endian PCM, forwards every byte unchanged, emits native
      `stt.signal` detections with `origin=router`, and keeps the following
      commit acknowledgement separate. Deepgram Flux has no router-VAD path —
      its client grammar carries no commit frame — so it serves semantic turn
      detection only and rejects `turn_detection=acoustic`.

      Portable input mapping:

      | Provider/model family | Default | Native activation/translation |
      |---|---|---|
      | Deepgram Nova | acoustic | `endpointing`, `utterance_end_ms`, `vad_events` |
      | Deepgram Flux | semantic (only) | `eot_threshold`, `eager_eot_threshold`, `eot_timeout_ms` |
      | ElevenLabs Scribe realtime | acoustic | `commit_strategy=vad`, `vad_*`, `min_*` |
      | Soniox realtime | acoustic | `enable_endpoint_detection`, `max_endpoint_delay_ms` |
      | AssemblyAI streaming | semantic turns | model-specific turn silence/confidence fields |
      | xAI Grok | acoustic; semantic on request | endpointing or Smart Turn |
      | Cartesia ink-2 | semantic turns | `stream_mode=turns` |
      | Cartesia ink-whisper | acoustic | router VAD; no provider turn detector |
      | OpenAI transcription | acoustic | nested server-VAD session config |
      | OpenAI realtime Whisper | acoustic | router VAD plus native commit control |
      | Together | acoustic | `threshold`, `min_silence_duration_ms`, `min_speech_duration_ms` |

      References: [Deepgram](https://developers.deepgram.com/docs/endpointing),
      [ElevenLabs](https://elevenlabs.io/docs/api-reference/speech-to-text/realtime),
      [Soniox](https://soniox.com/docs/stt/api-reference/websocket-api),
      [AssemblyAI](https://www.assemblyai.com/docs/speech-to-text/universal-streaming),
      [xAI](https://docs.x.ai/docs/guides/speech-to-text),
      [Cartesia](https://docs.cartesia.ai/api-reference/stt/stt),
      [OpenAI](https://platform.openai.com/docs/guides/realtime-transcription),
      [Together](https://docs.together.ai/docs/inference/transcription/voice-activity-detection).

      Close-code semantics: `1000` normal close, `1011` provider session
      failure, and `4401`/`4403` authentication failures. Close reason text is
      diagnostic and may change; applications should branch on the close code.
    bindings:
      ws:
        method: GET
        query:
          type: object
          description: Portable handshake parameters. Unknown bare keys are rejected; provider-native variables belong under `provider_options[<name>]`.
          required:
            - model
          properties:
            provider:
              type: string
              description: Pin a specific STT provider (e.g. `soniox`, `assemblyai`, `deepgram`, `grok`).
            model:
              type: string
              description: Model id, bare (`nova-3`) or `{author}/{modelName}` slug (`deepgram/nova-3`).
            language:
              type: string
              description: Expected language.
            audio_format:
              type: string
              enum:
                - pcm_8000
                - pcm_16000
                - pcm_22050
                - pcm_24000
                - pcm_44100
                - pcm_48000
                - ulaw_8000
              description: Required declaration of the unchanged raw audio bytes.
            channels:
              type: integer
              default: 1
            commit_strategy:
              type: string
              enum:
                - auto
                - manual
              default: auto
            turn_detection:
              type: string
              enum:
                - auto
                - acoustic
                - semantic
              default: auto
            silence_duration_ms:
              type: number
              minimum: 0
            min_speech_duration_ms:
              type: number
              minimum: 0
            min_silence_duration_ms:
              type: number
              minimum: 0
            turn_timeout_ms:
              type: number
              minimum: 0
            speech_threshold:
              type: number
              minimum: 0
              maximum: 1
            turn_threshold:
              type: number
              minimum: 0
              maximum: 1
            eager_end_threshold:
              type: number
              minimum: 0
              maximum: 1
            interim_results:
              type: boolean
              description: "Request mutable transcript hypotheses. There is no global default: when omitted, the selected model family's own behavior applies (`false` for Deepgram Flux, `true` elsewhere). The resolved value is echoed in `stt.session.started.input_config.interim_results`."
            include_timestamps:
              type: boolean
            include_language_detection:
              type: boolean
            keyterms:
              type: array
              items:
                type: string
            context:
              type: string
            diarization:
              type: boolean
            formatting:
              type: boolean
            include_filler_words:
              type: boolean
            event_format:
              type: string
              enum:
                - normalized
                - provider
              default: normalized
              description: "`normalized` emits the portable sequenced `stt.*` contract. `provider` forwards the selected provider's frames without mapping, rewriting, deduplication, or router-generated JSON events and expects provider-native text controls. It requires a direct upstream realtime WebSocket."
            provider_order:
              type: string
              description: Comma-separated or repeated provider ids in preferred order.
            provider_only:
              type: string
              description: Comma-separated or repeated provider ids; only these may serve the request.
            provider_ignore:
              type: string
              description: Comma-separated or repeated provider ids that must not serve the request.
            allow_fallbacks:
              type: string
              description: Set to `false` to prevent fallback to another provider.
          additionalProperties: true
        headers:
          type: object
          description: Handshake headers (besides the standard WebSocket upgrade headers).
          properties:
            authorization:
              type: string
              description: AllModels API key as `Bearer <key>`.
        bindingVersion: 0.1.0
    messages:
      nativeSttAudioChunk:
        $ref: "#/components/messages/nativeSttAudioChunk"
      nativeSttClientControlFrame:
        $ref: "#/components/messages/nativeSttClientControlFrame"
      nativeSttProviderEvent:
        $ref: "#/components/messages/nativeSttProviderEvent"
      nativeSttEvent:
        $ref: "#/components/messages/nativeSttEvent"
  elevenlabsTtsStreamInput:
    address: /el/v1/text-to-speech/{voice_id}/stream-input
    title: ElevenLabs SDK-compatible streaming-input TTS
    summary: ElevenLabs stream-input protocol across supported TTS providers.
    description: |
      Implements ElevenLabs' `stream-input` WebSocket protocol. All frames in
      both directions are JSON text frames; audio is carried as base64 inside
      the JSON (unlike `/v1/tts`, which uses binary frames).

      Handshake: HTTP GET with `Upgrade: websocket`, authenticated with
      `xi-api-key` or `Authorization: Bearer`. Select the provider
      with a canonical `{provider}/{model}` `model_id`, or use bare-model
      inference when the model maps unambiguously to one provider.

      Text is buffered by default to improve prosody across message boundaries.
      `try_trigger_generation` and `flush` are supported by
      ElevenLabs, Deepgram, Grok, and Fish; MiniMax cannot flush mid-stream.
      `auto_mode` (query param) and `generation_config.chunk_length_schedule`
      (init frame) are ElevenLabs-only buffering controls.

      Deepgram WebSocket TTS emits only `pcm_*`/`ulaw_*`; an `mp3_*`
      `output_format` against Deepgram is rejected before the upgrade with
      `400 invalid_output_format`.

      Close-code semantics: `1000` normal completion, `1011` provider session
      failure, and `4401`/`4403` authentication failures. Close reason text is
      diagnostic and may change.
    parameters:
      voice_id:
        description: Provider voice id (path segment), e.g. `eve` for Grok TTS. Forwarded to the routed provider verbatim — the router applies no default voice and recognises no placeholder; see `defaults.tts.voice` on GET /v1/providers for a recommended value per provider.
    bindings:
      ws:
        method: GET
        query:
          type: object
          description: Handshake query parameters. `provider_options[<name>]=<value>` bracket params are also accepted and take precedence over bare keys. Recognized enum and boolean option values are case-insensitive and are normalized to each provider's wire spelling; free-form values such as prompts, keyterms, and voice IDs retain their original case.
          required:
            - model_id
          properties:
            model_id:
              type: string
              description: Model id, bare or `{author}/{modelName}` slug (`elevenlabs/eleven_flash_v2_5`).
            output_format:
              type: string
              description: Provider output format, e.g. `mp3_44100_128`, `pcm_16000`, or `ulaw_8000`.
            inactivity_timeout:
              type: string
              description: Seconds of client inactivity before the ElevenLabs connection is closed.
            auto_mode:
              type: string
              description: ElevenLabs only. Generate audio per message instead of buffering text across messages.
            enable_logging:
              type: string
              description: ElevenLabs only. Set to `false` to disable provider request logging.
            language_code:
              type: string
              description: ISO language code for speech generation.
            apply_text_normalization:
              type: string
              enum:
                - auto
                - on
                - off
              description: ElevenLabs text-normalization mode.
            enable_ssml_parsing:
              type: string
              description: ElevenLabs only. Parse SSML tags in the input text.
            provider_order:
              type: string
              description: Comma-separated or repeated provider ids in preferred order.
            provider_only:
              type: string
              description: Comma-separated or repeated provider ids; only these may serve the request.
            provider_ignore:
              type: string
              description: Comma-separated or repeated provider ids that must not serve the request.
            allow_fallbacks:
              type: string
              description: Set to `false` to prevent fallback to another provider.
          additionalProperties: true
        headers:
          type: object
          description: Handshake headers (besides the standard WebSocket upgrade headers).
          properties:
            xi-api-key:
              type: string
              description: AllModels API key (ElevenLabs SDK convention).
            authorization:
              type: string
              description: "`Bearer <key>` alternative to `xi-api-key`."
        bindingVersion: 0.1.0
    messages:
      elTtsInit:
        $ref: "#/components/messages/elTtsInit"
      elTtsTextChunk:
        $ref: "#/components/messages/elTtsTextChunk"
      elTtsFlush:
        $ref: "#/components/messages/elTtsFlush"
      elTtsClose:
        $ref: "#/components/messages/elTtsClose"
      elTtsAudioChunk:
        $ref: "#/components/messages/elTtsAudioChunk"
      elTtsFinal:
        $ref: "#/components/messages/elTtsFinal"
      elTtsError:
        $ref: "#/components/messages/elTtsError"
  elevenlabsSttRealtime:
    address: /el/v1/speech-to-text/realtime
    title: ElevenLabs SDK-compatible realtime STT
    summary: ElevenLabs realtime STT protocol across supported STT providers.
    description: |
      Implements the ElevenLabs realtime speech-to-text protocol
      (`client.speechToText.realtime.connect(...)`). All frames in both
      directions are JSON text frames discriminated by `message_type`; audio
      is carried base64-encoded inside `input_audio_chunk` frames.

      Handshake: HTTP GET with `Upgrade: websocket`, authenticated with
      `xi-api-key` or `Authorization: Bearer`. The ElevenLabs SDK does not
      forward custom headers on this handshake, so select a provider with a
      `{provider}/{model}` `model_id`.

      Commit strategy: `commit_strategy=vad` uses server-side voice
      activity detection to find utterance boundaries (tune via the `vad_*` /
      `min_*` params); `commit_strategy=manual` lets the client close an
      utterance explicitly by sending an `input_audio_chunk` frame with
      `commit: true` (the SDK's `conn.commit()`) and is the official-default
      behavior preserved by this compatibility surface.

      Close-code semantics: `1000` normal close, `1011` provider session
      failure, and `4401`/`4403` authentication failures. Protocol errors that keep the
      socket open are reported as JSON frames whose `message_type` carries the
      error code. Close reason text is diagnostic and may change.
    bindings:
      ws:
        method: GET
        query:
          type: object
          description: Handshake query parameters. `provider_options[<name>]=<value>` bracket params are also accepted and take precedence over bare keys. Recognized enum and boolean option values are case-insensitive and are normalized to each provider's wire spelling; free-form values such as prompts, keyterms, and voice IDs retain their original case.
          required:
            - model_id
          properties:
            model_id:
              type: string
              description: Model id, bare or `{author}/{modelName}` slug (`deepgram/nova-3`).
            language_code:
              type: string
              description: Expected language.
            audio_format:
              type: string
              description: Input audio format, such as `pcm_16000` or `ulaw_8000`.
            keyterms:
              type: string
              description: Keyword/keyterm biasing. Repeat the param for multiple terms.
            vad_silence_threshold_secs:
              type: string
              description: "VAD tuning (auto commit): trailing silence, in seconds, that ends an utterance."
            vad_threshold:
              type: string
              description: "VAD tuning (auto commit): speech-probability threshold (0-1) for detecting speech."
            min_speech_duration_ms:
              type: string
              description: "VAD tuning (auto commit): minimum speech length, in ms, before an utterance starts."
            min_silence_duration_ms:
              type: string
              description: "VAD tuning (auto commit): minimum silence length, in ms, before an utterance ends."
            include_timestamps:
              type: string
              description: When set, committed transcripts include word/char timestamps (`committed_transcript_with_timestamps` frames).
            include_language_detection:
              type: string
              description: Request language detection metadata where supported.
            commit_strategy:
              type: string
              enum:
                - vad
                - manual
              default: manual
              description: Utterance segmentation; see the channel description.
            no_verbatim:
              type: string
              description: When set, requests non-verbatim (cleaned/formatted) transcripts where the provider supports it.
            provider_order:
              type: string
              description: Comma-separated or repeated provider ids in preferred order.
            provider_only:
              type: string
              description: Comma-separated or repeated provider ids; only these may serve the request.
            provider_ignore:
              type: string
              description: Comma-separated or repeated provider ids that must not serve the request.
            allow_fallbacks:
              type: string
              description: Set to `false` to prevent fallback to another provider.
          additionalProperties: true
        headers:
          type: object
          description: Handshake headers (besides the standard WebSocket upgrade headers).
          properties:
            xi-api-key:
              type: string
              description: AllModels API key (ElevenLabs SDK convention).
            authorization:
              type: string
              description: "`Bearer <key>` alternative to `xi-api-key`."
        bindingVersion: 0.1.0
    messages:
      elSttInputAudioChunk:
        $ref: "#/components/messages/elSttInputAudioChunk"
      elSttSessionStarted:
        $ref: "#/components/messages/elSttSessionStarted"
      elSttPartialTranscript:
        $ref: "#/components/messages/elSttPartialTranscript"
      elSttCommittedTranscript:
        $ref: "#/components/messages/elSttCommittedTranscript"
      elSttCommittedTranscriptWithTimestamps:
        $ref: "#/components/messages/elSttCommittedTranscriptWithTimestamps"
      elSttError:
        $ref: "#/components/messages/elSttError"
  openaiRealtime:
    address: /oai/realtime
    title: OpenAI SDK-compatible realtime transcription
    summary: GA OpenAI Realtime transcription events across supported STT providers.
    description: |
      Implements the **GA transcription subset** of the OpenAI Realtime
      protocol — session configuration, input audio buffering, and input-audio
      transcription events. Conversation items, responses, output audio, and
      WebRTC are not part of this surface. All frames in both directions are
      JSON text frames discriminated by `type`.

      Speech-activity events are emitted only when the session enables a server
      turn detector **and** the selected provider reports speech activity (for
      example, Deepgram Nova with `vad_events`, AssemblyAI, or OpenAI realtime).
      Transcript-only providers emit none; their absence is not an error. With
      `turn_detection: null` (or type `none`) the client owns utterance
      boundaries, so `input_audio_buffer.speech_started` /
      `input_audio_buffer.speech_stopped` are never emitted — provider activity
      or turn signals are suppressed even for providers that still report them.

      Handshake: HTTP GET with `Upgrade: websocket`. Node/Python OpenAI SDK
      clients authenticate with `Authorization: Bearer <key>`. Browser clients
      may instead offer the WebSocket subprotocols
      `realtime, openai-insecure-api-key.<key>`. When `realtime` is offered,
      AllModels returns `sec-websocket-protocol: realtime` on the 101 response.

      Send `session.update` (or `transcription_session.update`) before audio
      to choose the model, language, audio format, turn detection, and — as
      AllModels extensions — `session.provider` and `session.provider_options`.

      **Session lock:** once `input_audio_buffer.append` audio has been
      received, connection-level configuration is locked. A later session
      update that changes the provider, the transcription model, the input
      audio format, the commit strategy (turn detection on/off), or a locked
      `provider_options` connection key (`audio_format`, `sample_rate`,
      `encoding`, `commit_strategy`) is rejected with an `error` event whose
      `error.code` is `session_locked`. Compatible updates remain allowed.

      Close-code semantics: `1000` normal close, `1011` provider session
      failure, and `4401`/`4403` authentication failures. Close reason text is
      diagnostic and may change.
    bindings:
      ws:
        method: GET
        query:
          type: object
          description: Handshake query parameters. `provider_options[<name>]=<value>` bracket params are also accepted; body/session extensions take precedence over query controls. Recognized enum and boolean option values are case-insensitive and are normalized to each provider's wire spelling; free-form values such as prompts, keyterms, and voice IDs retain their original case.
          required:
            - model
          properties:
            model:
              type: string
              description: Realtime transcription model, preferably as a `{author}/{modelName}` slug.
            provider_order:
              type: string
              description: Comma-separated or repeated provider ids in preferred order.
            provider_only:
              type: string
              description: Comma-separated or repeated provider ids; only these may serve the request.
            provider_ignore:
              type: string
              description: Comma-separated or repeated provider ids that must not serve the request.
            allow_fallbacks:
              type: string
              description: Set to `false` to prevent fallback to another provider.
          additionalProperties: true
        headers:
          type: object
          description: Handshake headers (besides the standard WebSocket upgrade headers).
          properties:
            authorization:
              type: string
              description: "`Bearer <key>` — the standard OpenAI SDK credential slot."
            sec-websocket-protocol:
              type: string
              description: "Browser auth alternative: offer `realtime` plus `openai-insecure-api-key.<key>`. AllModels echoes `realtime` back when offered."
        bindingVersion: 0.1.0
    messages:
      oaiSessionUpdate:
        $ref: "#/components/messages/oaiSessionUpdate"
      oaiTranscriptionSessionUpdate:
        $ref: "#/components/messages/oaiTranscriptionSessionUpdate"
      oaiInputAudioBufferAppend:
        $ref: "#/components/messages/oaiInputAudioBufferAppend"
      oaiInputAudioBufferCommit:
        $ref: "#/components/messages/oaiInputAudioBufferCommit"
      oaiInputAudioBufferClear:
        $ref: "#/components/messages/oaiInputAudioBufferClear"
      oaiClose:
        $ref: "#/components/messages/oaiClose"
      oaiSessionCreated:
        $ref: "#/components/messages/oaiSessionCreated"
      oaiSessionUpdated:
        $ref: "#/components/messages/oaiSessionUpdated"
      oaiTranscriptionSessionUpdated:
        $ref: "#/components/messages/oaiTranscriptionSessionUpdated"
      oaiInputAudioBufferSpeechStarted:
        $ref: "#/components/messages/oaiInputAudioBufferSpeechStarted"
      oaiInputAudioBufferSpeechStopped:
        $ref: "#/components/messages/oaiInputAudioBufferSpeechStopped"
      oaiInputAudioBufferCommitted:
        $ref: "#/components/messages/oaiInputAudioBufferCommitted"
      oaiInputAudioBufferCleared:
        $ref: "#/components/messages/oaiInputAudioBufferCleared"
      oaiTranscriptionDelta:
        $ref: "#/components/messages/oaiTranscriptionDelta"
      oaiTranscriptionCompleted:
        $ref: "#/components/messages/oaiTranscriptionCompleted"
      oaiError:
        $ref: "#/components/messages/oaiError"
  speechEngineConversation:
    address: /el/v1/convai/conversation
    title: ElevenLabs-compatible Speech Engine conversation
    summary: Direct client STT, application response, and streamed TTS conversation.
    description: |
      Obtain `signed_url` from the authenticated HTTP
      `/el/v1/convai/conversation/get-signed-url` endpoint, then connect the
      client device directly to this WebSocket URL. Signed URLs are short-lived
      and single-use. Audio sent immediately after the WebSocket opens is
      accepted while the session starts.

      Audio may be binary or an ElevenLabs `user_audio_chunk` base64 JSON frame.
      Synthesized audio is returned as `audio` JSON frames with base64 content.
      Your configured application WebSocket receives transcripts and streams
      response text back to AllModels. AllModels includes a signed authorization
      token when connecting to your application.

      Conversation content and audio are not persisted.

      Beyond the official ElevenLabs frames this channel also emits AllModels
      extension frames whose `type` is prefixed `allmodels_`
      (`allmodels_activity_started`, `allmodels_activity_stopped`,
      `allmodels_transcript_delta`, `allmodels_audio_complete`, and
      `allmodels_latency`). Some are opt-in; each message documents its own
      condition. Clients that implement only the ElevenLabs protocol can
      ignore every `allmodels_`-prefixed frame safely.

      Errors arrive as an `error` frame (not `client_error`). Session-fatal
      errors are followed by a close: `4401` authentication, `4402` balance,
      and `4403` for every other access or runtime failure. A per-turn
      application failure (`error_name: application_response_failed`) does not
      close the socket — only the response in flight is abandoned.

      The customer application WebSocket is a separate leg and is not modeled
      by this channel; its frames — including the per-turn
      `{"type":"error","event_id":<n>,"message":"<detail>"}` frame an
      application can send to fail one response — are documented in
      `docs/speech-engine.md`.
    bindings:
      ws:
        method: GET
        query:
          type: object
          required:
            - conversation_signature
          properties:
            conversation_signature:
              type: string
              description: Opaque short-lived grant returned inside `signed_url`.
            agent_id:
              type: string
              description: Optional Speech Engine id included by compatible clients.
        bindingVersion: 0.1.0
    messages:
      speechEngineAudioInput:
        $ref: "#/components/messages/speechEngineAudioInput"
      speechEnginePong:
        $ref: "#/components/messages/speechEnginePong"
      speechEngineUserMessage:
        $ref: "#/components/messages/speechEngineUserMessage"
      speechEngineMetadata:
        $ref: "#/components/messages/speechEngineMetadata"
      speechEngineUserTranscript:
        $ref: "#/components/messages/speechEngineUserTranscript"
      speechEngineAgentResponse:
        $ref: "#/components/messages/speechEngineAgentResponse"
      speechEngineAudioOutput:
        $ref: "#/components/messages/speechEngineAudioOutput"
      speechEnginePing:
        $ref: "#/components/messages/speechEnginePing"
      speechEngineInterruption:
        $ref: "#/components/messages/speechEngineInterruption"
      speechEngineResponseComplete:
        $ref: "#/components/messages/speechEngineResponseComplete"
      speechEngineActivityStarted:
        $ref: "#/components/messages/speechEngineActivityStarted"
      speechEngineActivityStopped:
        $ref: "#/components/messages/speechEngineActivityStopped"
      speechEngineTranscriptDelta:
        $ref: "#/components/messages/speechEngineTranscriptDelta"
      speechEngineAudioComplete:
        $ref: "#/components/messages/speechEngineAudioComplete"
      speechEngineLatency:
        $ref: "#/components/messages/speechEngineLatency"
      speechEngineClientError:
        $ref: "#/components/messages/speechEngineClientError"
  nativeRealtime:
    address: /v1/realtime
    title: AllModels Realtime voice session
    summary: Direct client audio, normalized transcription, application text, and synthesized audio.
    description: |
      Connect with a single-use `client_secret` returned by
      `POST /v1/realtime/client-secrets`, or with `profile_id` and an
      `Authorization: Bearer <key>` header. Client audio sent immediately after
      the WebSocket opens is accepted while the session starts. Binary client
      frames contain raw input audio; binary server frames contain synthesized
      audio unless the profile selects base64 transport.
    bindings:
      ws:
        method: GET
        query:
          type: object
          properties:
            client_secret:
              type: string
              description: Single-use, short-lived client secret.
            profile_id:
              type: string
              description: Realtime profile id for server-side API-key connections.
        bindingVersion: 0.1.0
    messages:
      nativeRealtimeClientEvent:
        $ref: "#/components/messages/nativeRealtimeClientEvent"
      nativeRealtimeServerEvent:
        $ref: "#/components/messages/nativeRealtimeServerEvent"
operations:
  nativeTtsClientMessages:
    action: receive
    channel:
      $ref: "#/channels/nativeTts"
    security:
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Native TTS client frames
    summary: JSON command frames the client sends on /v1/tts.
    messages:
      - $ref: "#/channels/nativeTts/messages/nativeTtsText"
      - $ref: "#/channels/nativeTts/messages/nativeTtsFlush"
      - $ref: "#/channels/nativeTts/messages/nativeTtsClose"
  nativeTtsServerMessages:
    action: send
    channel:
      $ref: "#/channels/nativeTts"
    security:
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Native TTS server frames
    summary: Session status and binary audio returned on /v1/tts.
    messages:
      - $ref: "#/channels/nativeTts/messages/sessionReady"
      - $ref: "#/channels/nativeTts/messages/nativeTtsAudioChunk"
      - $ref: "#/channels/nativeTts/messages/nativeTtsSessionEnded"
      - $ref: "#/channels/nativeTts/messages/nativeTtsError"
      - $ref: "#/channels/nativeTts/messages/nativeTtsProviderStatus"
      - $ref: "#/channels/nativeTts/messages/nativeTtsPolicyCheck"
  nativeSttClientMessages:
    action: receive
    channel:
      $ref: "#/channels/nativeStt"
    security:
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Native STT client frames
    summary: Binary audio and normalized controls sent on /v1/stt.
    messages:
      - $ref: "#/channels/nativeStt/messages/nativeSttAudioChunk"
      - $ref: "#/channels/nativeStt/messages/nativeSttClientControlFrame"
  nativeSttServerMessages:
    action: send
    channel:
      $ref: "#/channels/nativeStt"
    security:
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Native STT server frames
    summary: Normalized events, or unmodified provider frames with event_format=provider.
    messages:
      - $ref: "#/channels/nativeStt/messages/nativeSttProviderEvent"
      - $ref: "#/channels/nativeStt/messages/nativeSttEvent"
  elevenlabsTtsStreamInputClientMessages:
    action: receive
    channel:
      $ref: "#/channels/elevenlabsTtsStreamInput"
    security:
      - $ref: "#/components/securitySchemes/xiApiKey"
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Stream-input TTS client frames
    summary: ElevenLabs stream-input JSON frames the client sends.
    messages:
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsInit"
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsTextChunk"
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsFlush"
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsClose"
  elevenlabsTtsStreamInputServerMessages:
    action: send
    channel:
      $ref: "#/channels/elevenlabsTtsStreamInput"
    security:
      - $ref: "#/components/securitySchemes/xiApiKey"
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Stream-input TTS server frames
    summary: Base64 audio chunks, terminal completion, and error frames returned by AllModels.
    messages:
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsAudioChunk"
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsFinal"
      - $ref: "#/channels/elevenlabsTtsStreamInput/messages/elTtsError"
  elevenlabsSttRealtimeClientMessages:
    action: receive
    channel:
      $ref: "#/channels/elevenlabsSttRealtime"
    security:
      - $ref: "#/components/securitySchemes/xiApiKey"
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Realtime STT client frames
    summary: ElevenLabs realtime STT frames the client sends.
    messages:
      - $ref: "#/channels/elevenlabsSttRealtime/messages/elSttInputAudioChunk"
  elevenlabsSttRealtimeServerMessages:
    action: send
    channel:
      $ref: "#/channels/elevenlabsSttRealtime"
    security:
      - $ref: "#/components/securitySchemes/xiApiKey"
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: Realtime STT server frames
    summary: ElevenLabs-compatible transcript events returned by AllModels.
    messages:
      - $ref: "#/channels/elevenlabsSttRealtime/messages/elSttSessionStarted"
      - $ref: "#/channels/elevenlabsSttRealtime/messages/elSttPartialTranscript"
      - $ref: "#/channels/elevenlabsSttRealtime/messages/elSttCommittedTranscript"
      - $ref: "#/channels/elevenlabsSttRealtime/messages/elSttCommittedTranscriptWithTimestamps"
      - $ref: "#/channels/elevenlabsSttRealtime/messages/elSttError"
  openaiRealtimeClientMessages:
    action: receive
    channel:
      $ref: "#/channels/openaiRealtime"
    security:
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: OpenAI realtime client events
    summary: GA transcription client events the client sends on /oai/realtime.
    messages:
      - $ref: "#/channels/openaiRealtime/messages/oaiSessionUpdate"
      - $ref: "#/channels/openaiRealtime/messages/oaiTranscriptionSessionUpdate"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferAppend"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferCommit"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferClear"
      - $ref: "#/channels/openaiRealtime/messages/oaiClose"
  openaiRealtimeServerMessages:
    action: send
    channel:
      $ref: "#/channels/openaiRealtime"
    security:
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: OpenAI realtime server events
    summary: GA transcription server events returned on /oai/realtime.
    messages:
      - $ref: "#/channels/openaiRealtime/messages/oaiSessionCreated"
      - $ref: "#/channels/openaiRealtime/messages/oaiSessionUpdated"
      - $ref: "#/channels/openaiRealtime/messages/oaiTranscriptionSessionUpdated"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferSpeechStarted"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferSpeechStopped"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferCommitted"
      - $ref: "#/channels/openaiRealtime/messages/oaiInputAudioBufferCleared"
      - $ref: "#/channels/openaiRealtime/messages/oaiTranscriptionDelta"
      - $ref: "#/channels/openaiRealtime/messages/oaiTranscriptionCompleted"
      - $ref: "#/channels/openaiRealtime/messages/oaiError"
  speechEngineConversationClientMessages:
    action: receive
    channel:
      $ref: "#/channels/speechEngineConversation"
    security:
      - $ref: "#/components/securitySchemes/conversationSignature"
    title: Speech Engine client frames
    summary: Audio, pong, and optional typed user messages sent by a client device.
    messages:
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineAudioInput"
      - $ref: "#/channels/speechEngineConversation/messages/speechEnginePong"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineUserMessage"
  speechEngineConversationServerMessages:
    action: send
    channel:
      $ref: "#/channels/speechEngineConversation"
    security:
      - $ref: "#/components/securitySchemes/conversationSignature"
    title: Speech Engine server frames
    summary: Conversation metadata, transcripts, response text/audio, lifecycle, AllModels extensions, and errors.
    messages:
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineMetadata"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineUserTranscript"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineAgentResponse"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineAudioOutput"
      - $ref: "#/channels/speechEngineConversation/messages/speechEnginePing"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineInterruption"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineResponseComplete"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineActivityStarted"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineActivityStopped"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineTranscriptDelta"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineAudioComplete"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineLatency"
      - $ref: "#/channels/speechEngineConversation/messages/speechEngineClientError"
  nativeRealtimeClientMessages:
    action: receive
    channel:
      $ref: "#/channels/nativeRealtime"
    security:
      - $ref: "#/components/securitySchemes/realtimeClientSecret"
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: AllModels Realtime client events
    summary: Audio, commit, cancellation, heartbeat, and close events.
    messages:
      - $ref: "#/channels/nativeRealtime/messages/nativeRealtimeClientEvent"
  nativeRealtimeServerMessages:
    action: send
    channel:
      $ref: "#/channels/nativeRealtime"
    security:
      - $ref: "#/components/securitySchemes/realtimeClientSecret"
      - $ref: "#/components/securitySchemes/bearerAuth"
    title: AllModels Realtime server events
    summary: Session, speech, transcript, response, metric, heartbeat, and error events.
    messages:
      - $ref: "#/channels/nativeRealtime/messages/nativeRealtimeServerEvent"
components:
  securitySchemes:
    xiApiKey:
      type: httpApiKey
      name: xi-api-key
      in: header
      description: AllModels API key (ElevenLabs SDK convention).
    bearerAuth:
      type: http
      scheme: bearer
      description: AllModels API key supplied as the Authorization Bearer token.
    conversationSignature:
      type: httpApiKey
      name: conversation_signature
      in: query
      description: Opaque grant returned in the signed conversation URL.
    realtimeClientSecret:
      type: httpApiKey
      name: client_secret
      in: query
      description: Single-use client secret returned by POST /v1/realtime/client-secrets.
  messages:
    nativeTtsText:
      name: nativeTtsText
      title: Text frame
      summary: Append text to be synthesized.
      examples:
        - name: appendText
          payload:
            type: text
            text: Hello from AllModels.
      payload:
        type: object
        properties:
          type:
            type: string
            const: text
          text:
            type: string
            maxLength: 2000
            description: Text to synthesize. Frames over 2000 characters close the socket with `1009`.
        required:
          - type
          - text
    nativeTtsFlush:
      name: nativeTtsFlush
      title: Flush frame
      summary: Force generation of buffered text.
      examples:
        - name: flush
          payload:
            type: flush
      payload:
        type: object
        properties:
          type:
            type: string
            const: flush
        required:
          - type
    nativeTtsClose:
      name: nativeTtsClose
      title: Close frame
      summary: Finish the stream after all buffered text is synthesized.
      examples:
        - name: close
          payload:
            type: close
      payload:
        type: object
        properties:
          type:
            type: string
            const: close
        required:
          - type
    sessionReady:
      name: sessionReady
      title: Session ready
      summary: Emitted when the streaming session is ready.
      description: Confirms that the session is ready after authentication and pricing admission. `queued_messages`, when present, reports how many client messages were accepted while the session was starting. Other properties are informational and are not part of the public contract.
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: allmodels.bridge_ready
          queued_messages:
            type: integer
            description: Client messages accepted while the session was starting.
        additionalProperties: true
      examples:
        - name: sessionReady
          payload:
            type: allmodels.bridge_ready
            queued_messages: 2
    nativeTtsAudioChunk:
      name: nativeTtsAudioChunk
      title: Audio chunk (binary)
      summary: Decoded provider audio as a binary WebSocket frame.
      contentType: application/octet-stream
      description: Binary frames carrying the synthesized audio. With `?format=pcm` (or no format) the bytes are raw 16-bit little-endian mono PCM at the provider's native sample rate; with `?format=mp3` they are MP3 bytes for providers that can emit MP3 (providers that only stream raw PCM, e.g. Cartesia and Together, fall back to PCM).
      payload:
        type: string
        format: binary
        description: Raw audio bytes (not JSON).
      examples:
        - name: pcmChunk
          summary: A binary frame of raw PCM bytes (shown here as a placeholder string).
          payload: <binary audio bytes>
    nativeTtsSessionEnded:
      name: nativeTtsSessionEnded
      title: TTS session ended
      summary: Marks successful completion before the WebSocket closes normally.
      description: Emitted exactly once before a clean completion close. A clean upstream close (code `1000` or `1005`) synthesizes the terminal frame and is normalized to `1000 / tts_complete`; abnormal upstream closes propagate their original code and reason with no terminal frame. `billed_characters` is present only when the provider reported an authoritative billed-character count.
      payload:
        type: object
        required:
          - type
          - reason
        properties:
          type:
            const: allmodels.tts_session_ended
          reason:
            const: tts_complete
          billed_characters:
            type: integer
            minimum: 0
            description: Latest authoritative provider-reported billed-character count.
        additionalProperties: false
      examples:
        - name: streamComplete
          payload:
            type: allmodels.tts_session_ended
            reason: tts_complete
        - name: streamCompleteWithUsage
          payload:
            type: allmodels.tts_session_ended
            reason: tts_complete
            billed_characters: 42
    nativeTtsError:
      name: nativeTtsError
      title: TTS error
      summary: Reports a normalized fatal provider error.
      description: Emitted before a fatal upstream failure closes the WebSocket with `1011 / upstream_error`. Provider error bodies are not forwarded verbatim.
      payload:
        type: object
        required:
          - type
          - message
        properties:
          type:
            const: error
          message:
            type: string
          code:
            type: string
        additionalProperties: false
      examples:
        - name: upstreamError
          payload:
            type: error
            message: Provider failed.
            code: upstream_error
    nativeTtsProviderStatus:
      name: nativeTtsProviderStatus
      title: Provider status
      summary: Carries a benign provider-specific status inside a stable envelope.
      description: 'The `payload` object is intentionally open because informational status fields vary by provider. Unexpected non-JSON adapter output is retained as `{"raw": "..."}`.'
      payload:
        type: object
        required:
          - type
          - payload
        properties:
          type:
            const: allmodels.provider_status
          payload:
            type: object
            additionalProperties: true
        additionalProperties: false
      examples:
        - name: providerMetadata
          payload:
            type: allmodels.provider_status
            payload:
              type: Metadata
    nativeTtsPolicyCheck:
      name: nativeTtsPolicyCheck
      title: Policy check
      summary: Reports runtime pricing and balance-policy timing.
      description: Best-effort telemetry emitted after each completed runtime credit-policy check. Optional timing fields appear only when that policy layer ran.
      payload:
        type: object
        required:
          - type
          - total_ms
          - pricing_ms
          - outcome
        properties:
          type:
            const: allmodels.policy_check
          total_ms:
            type: number
            minimum: 0
          pricing_ms:
            type: number
            minimum: 0
          org_balance_ms:
            type: number
            minimum: 0
          guardrail_policy_ms:
            type: number
            minimum: 0
          guardrail_ms:
            type: number
            minimum: 0
          outcome:
            enum:
              - allowed
              - insufficient_balance
              - quota_exceeded
              - unpriced
        additionalProperties: false
      examples:
        - name: allowedPolicyCheck
          payload:
            type: allmodels.policy_check
            total_ms: 8
            pricing_ms: 1
            org_balance_ms: 3
            guardrail_policy_ms: 2
            guardrail_ms: 2
            outcome: allowed
    nativeSttAudioChunk:
      name: nativeSttAudioChunk
      title: Audio chunk (binary)
      summary: Raw audio sent for transcription.
      contentType: application/octet-stream
      description: "Binary frames carrying input audio. The encoding and sample rate are declared by the single required handshake parameter `audio_format` (e.g. `audio_format=pcm_16000`). There is no separate `sample_rate` parameter — bare query keys outside the portable set are rejected with `unknown_stt_parameter`, and a native rate override under `provider_options[...]` that disagrees with `audio_format` is rejected with `conflicting_audio_format`. Accepted formats are model-specific: most streaming models take `pcm_8000`-`pcm_48000` and `ulaw_8000`, Together takes `pcm_16000` or `pcm_24000`, and OpenAI realtime transcription takes `pcm_24000` only."
      payload:
        type: string
        format: binary
        description: Raw audio bytes (not JSON).
      examples:
        - name: pcmChunk
          summary: A binary frame of raw PCM bytes (shown here as a placeholder string).
          payload: <binary audio bytes>
    nativeSttClientControlFrame:
      name: nativeSttClientControlFrame
      title: STT control frame
      summary: Portable native control, or a provider-native command in provider mode.
      description: |
        Normalized mode accepts `stt.audio.commit`, `stt.audio.clear`, and
        `stt.session.close`. With `event_format=provider`, text frames retain
        the selected provider's native grammar.
      payload:
        description: Provider-native command as a JSON object or plain string.
        anyOf:
          - type: string
          - type: object
            required:
              - type
            properties:
              type:
                enum:
                  - stt.audio.commit
                  - stt.audio.clear
                  - stt.session.close
                  - finalize
                  - ForceEndpoint
                  - Terminate
                  - audio.done
                  - Finalize
                  - CloseStream
                  - commit
                  - close
            additionalProperties: true
      examples:
        - name: commit
          payload:
            type: stt.audio.commit
        - name: clear
          payload:
            type: stt.audio.clear
        - name: close
          payload:
            type: stt.session.close
    nativeSttProviderEvent:
      name: nativeSttProviderEvent
      title: Provider-native transcript event
      summary: Transcript or session event in the selected provider's native format.
      description: |
        Soniox, AssemblyAI, Deepgram, Grok, Cartesia, Together, OpenAI, Gemini
        Live, and ElevenLabs return their native event formats. The following
        list provides common discriminators; consult the provider's protocol
        reference for its complete schema:
        - **Soniox** — `{"tokens": [{"text", "is_final", "start_ms", "end_ms", ...}], ...}`.
          Reference: https://soniox.com/docs/stt/api-reference/websocket-api
        - **AssemblyAI** — frames discriminated by `type`
          (`Begin`, `Turn`, `Termination`; errors use `Error`). Reference:
          https://www.assemblyai.com/docs/speech-to-text/universal-streaming
        - **Deepgram** — frames discriminated by `type`: `Results` (with
          `channel.alternatives[].transcript` and `is_final`) and `Metadata`
          (request id, billed duration). Reference:
          https://developers.deepgram.com/docs/lower-level-websockets
        - **Grok (x.ai)** — frames discriminated by a `transcript.*` `type`
          (`transcript.created`, `transcript.partial`, `transcript.done`,
          ...). Reference: https://docs.x.ai/
        - **Cartesia** — `{"type": "transcript", "text", "is_final",
          "words", ...}` transcript frames plus `done` / `error` frames.
          Reference: https://docs.cartesia.ai/
        - **Together / OpenAI** — OpenAI-realtime-style events discriminated
          by `type` (`conversation.item.input_audio_transcription.delta`,
          `...completed`, ...). Reference:
          https://platform.openai.com/docs/guides/realtime
        - **Gemini Live** — messages carrying `setupComplete` or
          `serverContent` (with `inputTranscription.text` deltas and a
          `turnComplete` or, on the dedicated transcription Live model,
          `generationComplete` turn-end flag), sent as JSON over binary
          frames. Reference: https://ai.google.dev/api/live
        - **ElevenLabs** — ElevenLabs' own realtime STT frames.

        Gemini Live may return JSON in binary frames.

        The provider envelope is passed through unchanged except that any
        `words[]` array in a transcript frame is normalized to the standardized
        word shape (see `sttTranscriptWord`): each entry carries `text`,
        `start`, `end`, and a `type` such as `word`.
      payload:
        description: Provider-native event as a JSON object or plain string.
        anyOf:
          - type: string
          - type: object
            additionalProperties: true
            not:
              anyOf:
                - required:
                    - kind
                - required:
                    - event_id
                - required:
                    - type
                  properties:
                    type:
                      enum:
                        - session_started
                        - allmodels.bridge_ready
                        - stt.audio.commit
                        - stt.audio.clear
                        - stt.session.close
                        - finalize
                        - ForceEndpoint
                        - Terminate
                        - audio.done
                        - Finalize
                        - CloseStream
                        - commit
                        - close
      examples:
        - name: sonioxTokens
          payload:
            tokens:
              - text: Hello
                is_final: true
                start_ms: 100
                end_ms: 400
        - name: assemblyaiTurn
          payload:
            type: Turn
            transcript: hello
            end_of_turn: false
        - name: deepgramResults
          payload:
            type: Results
            is_final: true
            channel:
              alternatives:
                - transcript: hello
        - name: cartesiaTranscript
          payload:
            type: transcript
            text: hello world
            is_final: true
            words:
              - text: hello
                start: 0
                end: 0.5
                type: word
              - text: world
                start: 0.5
                end: 1
                type: word
        - name: openaiRealtimeDelta
          summary: OpenAI Realtime-style event returned by Together and OpenAI.
          payload:
            type: conversation.item.input_audio_transcription.delta
            delta: "hello "
    nativeSttSessionStarted:
      name: nativeSttSessionStarted
      title: Session started
      summary: Session-start event for providers without a native equivalent.
      description: Returned when the selected provider does not define its own session-start event, including Soniox, Deepgram, Cartesia, Together, OpenAI, and Gemini.
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: session_started
      examples:
        - name: sessionStarted
          payload:
            type: session_started
    nativeSttAllModelsEvent:
      name: nativeSttAllModelsEvent
      title: AllModels transcript event
      summary: "AllModels transcript event shape (currently unreachable: no served streaming model emits it)."
      description: AllModels transcript events discriminated by `kind`. Fish, their one emitter, is batch-only and takes no streaming placement, so no served binding produces this shape today; the schema is retained for the event contract.
      payload:
        oneOf:
          - title: Session started
            type: object
            required:
              - kind
            properties:
              kind:
                const: session_started
              sessionId:
                type: string
                description: Provider session id, when available.
              durationSeconds:
                type: number
                description: Audio duration reported by the provider, when available.
          - title: Partial transcript
            type: object
            required:
              - kind
              - text
            properties:
              kind:
                const: partial_transcript
              text:
                type: string
              words:
                type: array
                items:
                  $ref: "#/components/schemas/sttTranscriptWord"
          - title: Committed transcript
            type: object
            required:
              - kind
              - text
            properties:
              kind:
                const: committed_transcript
              text:
                type: string
              words:
                type: array
                items:
                  $ref: "#/components/schemas/sttTranscriptWord"
              languageCode:
                type: string
                description: Language tag for the committed segment, when the provider reports one.
          - title: Error
            type: object
            required:
              - kind
              - code
              - message
            properties:
              kind:
                const: error
              code:
                type: string
                description: Machine-readable error code.
              message:
                type: string
                description: Human-readable error message.
      examples:
        - name: sessionStarted
          payload:
            kind: session_started
            sessionId: sess_123
        - name: partial
          payload:
            kind: partial_transcript
            text: hello wor
        - name: committed
          payload:
            kind: committed_transcript
            text: hello world
            languageCode: en
            words:
              - text: hello
                start: 0
                end: 0.5
                type: word
              - text: world
                start: 0.5
                end: 1
                type: word
        - name: error
          payload:
            kind: error
            code: transcription_failed
            message: Transcription failed.
    nativeSttEvent:
      name: nativeSttEvent
      title: Normalized streaming STT event
      summary: Sequenced transcript, unified signal, control, or session event.
      description: |
        Default `/v1/stt` server envelope. Every provider or router endpointing
        signal uses `type: stt.signal` with one flat `state`: `started` supports
        barge-in, `end_candidate` supports speculative work, `resumed` cancels it,
        and `end` is the strongest configured response boundary. Optional
        `turn_scope` distinguishes an acoustic utterance from a modeled
        conversational turn and is unrelated to transcript-final `scope`. `text`
        is present only on `end_candidate`, where it is the accumulated text for
        act-without-state; `end` never repeats transcript `text` or `words`.
        `speaker_id` is reserved for future per-speaker signal attribution and is
        not part of the current event schema.

        Provider signal mappings:

        | Provider | Models/mode | Activation | Provider event | Canonical state | Turn scope | Basis | Notes |
        |---|---|---|---|---|---|---|---|
        | Deepgram | Nova family | `vad_events=true` | `SpeechStarted` | `started` | none | VAD | Speech activity only. |
        | Deepgram | Nova family | Endpointing enabled | `Results.speech_final` / `UtteranceEnd` | `end` | utterance | VAD + silence | Equivalent duplicate boundaries are collapsed. |
        | Deepgram | Flux | Default turn detection | `StartOfTurn` / `EndOfTurn` | `started` / `end` | turn | Semantic | Provider turn index is retained as `turn_id`. |
        | Deepgram | Flux | `eager_eot_threshold` | `EagerEndOfTurn` / `TurnResumed` | `end_candidate` / `resumed` | turn | Semantic | Resume cancels speculative response work. |
        | AssemblyAI | Streaming models | Default turn detection | `SpeechStarted` / first `Turn` | `started` | none | Provider turn detector | Activity starts are deduplicated per turn. |
        | AssemblyAI | Streaming models | Default turn detection | `Turn.end_of_turn=true` | `end` | turn | Linguistic/semantic + silence | Formatted duplicate finals are suppressed. |
        | xAI | grok-stt | Without `smart_turn` | `speech_final=true` | `end` | utterance | Acoustic + silence | Represents an utterance endpoint. |
        | xAI | grok-stt | `smart_turn=<threshold>` | `speech_final=true` | `end` | turn | Semantic + silence | End-of-turn confidence is preserved. |
        | Soniox | Realtime | `enable_endpoint_detection=true` | `<end>` | `end` | utterance | Provider endpoint | No separate activity lifecycle. |
        | Cartesia | Ink turns mode | `stream_mode=turns` | `turn.start/eager_end/resume/end` | `started/end_candidate/resumed/end` | turn | Provider turn detector | Manual mode remains transcript-only. |
        | OpenAI | Realtime transcription | `turn_detection.type=server_vad` | `speech_started` / `speech_stopped` | `started` / `end` | utterance | VAD + silence | Speech stop also finalizes the provider input segment. |
        | OpenAI | Realtime transcription | `turn_detection.type=semantic_vad` | `speech_started` / `speech_stopped` | `started` / `end` | turn | Semantic + silence | Wire event names remain speech activity events. |
        | OpenAI | Realtime transcription | `turn_detection=null` or type `none` | None | None | none | None | Manual commits do not manufacture activity or boundary signals. |
        | Router (local VAD) | routerVad-eligible models | `turn_detection=acoustic` on routerVad-eligible models | EnergyVad onset / endpoint | `started` / `end` | utterance | VAD | Router detection signals precede the separate router commit acknowledgement. |
        | ElevenLabs / Together | Current streaming models | None | None | None | none | None | Transcript commits and local buffering do not manufacture signals. |
        | Gemini | Live STT | None | None | None | none | None | `turnComplete` (or `generationComplete` on the dedicated transcription Live model) follows the router's activity-end commit and is not mapped as a provider signal. |
        | Fish Audio | Current streaming models | None | None | None | none | None | Transcript commits and local buffering do not manufacture signals. |
        | Cloudflare | Workers AI streaming STT | None | None | None | none | None | Transcript commits and local buffering do not manufacture signals. |
        | Alibaba Cloud Model Studio (Qwen) | `qwen3-asr-flash-realtime` | `turn_detection.type=server_vad` (default) | `input_audio_buffer.speech_started` / `.speech_stopped` | `started` / `end` | utterance | VAD + silence | Speech stop precedes the provider's own transcription commit for the same utterance. |
        | Alibaba Cloud Model Studio (Qwen) | `qwen3-asr-flash-realtime` | `turn_detection=null` (`commit_strategy=manual`) | None | None | none | None | Manual turns are closed by the client commit and manufacture no activity signals. |
        | Inworld | `inworld/inworld-stt-1` | Server VAD (default) | `speechStarted` / `speechStopped` | `started` / `end` | utterance | VAD + silence | The detector needs streamed silence, not a client that stops sending; set turn_timeout_ms or commit manually otherwise. |
        | Inworld | `inworld/inworld-stt-1` | `commit_strategy=manual` (`vadThreshold=0`) | None | None | none | None | Manual turns are closed by the client commit, which sends endTurn, and manufacture no activity signals. |
        | Sprag AI | `rhythm` | `turn_detection.type=server_vad` (default) | `input_audio_buffer.speech_started` / `.speech_stopped` | `started` / `end` | utterance | VAD + silence | Speech stop precedes the provider's own transcription commit for the same utterance. |
        | Sprag AI | `rhythm` | `turn_detection=null` (`commit_strategy=manual`) | None | None | none | None | Manual turns are closed by the client commit and manufacture no activity signals. |

        `provider_event` and `provider_metadata` preserve source-vendor evidence.
        Router VAD may emit `started`/`end` with `origin: router`, utterance scope,
        VAD basis, and a frame-derived audio clock. Detection remains distinct
        from segmentation: the subsequent `stt.audio.committed` is a separate ack,
        and transcript commits never manufacture `stt.signal`. Client commit/clear
        acknowledgements remain separate control events.

        The `capabilities` object echoed by `stt.session.started` is a contract,
        not a hint: every candidate signal is filtered against `states` and, when
        present, `turn_scopes`. A stripped signal does not advance the turn
        lifecycle. Empty `states` means the session never emits `stt.signal`, and
        `predictive` is true exactly when `states` includes `end_candidate`. With
        `commit_strategy=manual`, a provider frame acknowledging the commit is
        diverted into the commit buffer, so a commit can never manufacture a
        signal. `stt.session.started` also carries `input_config`: the complete
        effective portable input the router resolved, plus `commit_origin`.
      x-allmodels-provider-signal-mappings:
        source: src/providers/sttSignalCapabilities.ts
        note: The Markdown table above is freshness-tested against the router-owned mapping manifest.
      payload:
        type: object
        required:
          - type
          - event_id
          - sequence
          - provider
          - origin
        properties:
          type:
            enum:
              - stt.session.started
              - stt.session.ended
              - stt.audio.committed
              - stt.audio.cleared
              - stt.transcript.partial
              - stt.transcript.final
              - stt.signal
              - stt.error
          event_id:
            type: string
          sequence:
            type: integer
            minimum: 1
          provider:
            type: string
          model:
            type: string
          session_id:
            type: string
          turn_id:
            type: string
          origin:
            enum:
              - provider
              - client
              - router
          provider_event:
            type: string
          text:
            type: string
            description: Full partial snapshot, immutable final chunk, or accumulated `end_candidate` text.
          delta:
            type: string
            description: True provider delta; omitted for snapshot providers.
          scope:
            enum:
              - segment
              - utterance
          words:
            type: array
            items:
              $ref: "#/components/schemas/sttTranscriptWord"
          language_code:
            type: string
          state:
            $ref: "#/components/schemas/sttSignalFields/properties/state"
          turn_scope:
            $ref: "#/components/schemas/sttSignalFields/properties/turn_scope"
          basis:
            $ref: "#/components/schemas/sttSignalFields/properties/basis"
          audio_start_ms:
            $ref: "#/components/schemas/sttSignalFields/properties/audio_start_ms"
          audio_end_ms:
            $ref: "#/components/schemas/sttSignalFields/properties/audio_end_ms"
          confidence:
            $ref: "#/components/schemas/sttSignalFields/properties/confidence"
          capabilities:
            $ref: "#/components/schemas/sttSignalCapabilities"
          input_config:
            type: object
            description: "Effective portable input after resolution: every portable request field the session resolved to, plus `commit_origin`. Fields the client did not send and the resolver did not default are absent."
            additionalProperties: false
            properties:
              audio_format:
                type: string
                description: Declared raw input encoding and sample rate, unchanged from the handshake.
              channels:
                type: integer
                description: Declared input channel count; only mono (`1`) is accepted today.
              language:
                type: string
                description: Requested language, unchanged from the handshake.
              commit_strategy:
                enum:
                  - auto
                  - manual
                description: Resolved segmentation mode. Always present; defaults to `auto`.
              turn_detection:
                enum:
                  - acoustic
                  - semantic
                description: Detector actually selected for `commit_strategy=auto` (`auto` is already resolved to a concrete detector). Absent when `commit_strategy=manual`, which disables automatic detection.
              silence_duration_ms:
                type: number
                description: Requested trailing silence used for automatic segmentation, as translated to the provider's native endpointing parameter.
              min_speech_duration_ms:
                type: number
                description: Requested minimum speech duration before an utterance may start.
              min_silence_duration_ms:
                type: number
                description: Requested minimum silence duration before a boundary may be produced.
              turn_timeout_ms:
                type: number
                description: Requested maximum wait for a provider turn or endpoint decision.
              speech_threshold:
                type: number
                description: Requested speech/VAD probability threshold (0-1).
              turn_threshold:
                type: number
                description: Requested semantic end-of-turn confidence threshold (0-1).
              eager_end_threshold:
                type: number
                description: Requested predictive eager-end threshold (0-1); predictive boundaries stay opt-in.
              interim_results:
                type: boolean
                description: Resolved interim-transcript behavior. Always present; when the client did not ask, it is the selected model family's default (false for Deepgram Flux, true elsewhere).
              include_timestamps:
                type: boolean
                description: Requested provider word/character timestamps.
              include_language_detection:
                type: boolean
                description: Requested provider language-detection metadata.
              keyterms:
                type: array
                items:
                  type: string
                description: Requested speech-recognition keyterms, as translated to the provider's biasing parameter.
              context:
                type: string
                description: Requested recognition context or prompt.
              diarization:
                type: boolean
                description: Requested speaker diarization.
              formatting:
                type: boolean
                description: Requested provider-native transcript formatting.
              include_filler_words:
                type: boolean
                description: Requested filler/disfluency words.
              commit_origin:
                enum:
                  - provider
                  - router
                  - client
                description: "Who turns streamed audio into committed utterances: the provider's own detector, router VAD, or explicit client `stt.audio.commit` frames. Always present."
          code:
            type: string
          message:
            type: string
          provider_metadata:
            type: object
            additionalProperties: true
        oneOf:
          - title: Session started
            required:
              - capabilities
            properties:
              type:
                const: stt.session.started
          - title: Session ended or control lifecycle
            properties:
              type:
                enum:
                  - stt.session.ended
                  - stt.audio.committed
                  - stt.audio.cleared
          - title: Transcript update
            required:
              - text
            properties:
              type:
                enum:
                  - stt.transcript.partial
                  - stt.transcript.final
          - title: Unified provider signal
            required:
              - state
            properties:
              type:
                const: stt.signal
            allOf:
              - not:
                  required:
                    - words
              - if:
                  properties:
                    state:
                      const: end_candidate
                  required:
                    - state
                else:
                  not:
                    required:
                      - text
          - title: Error
            required:
              - code
              - message
            properties:
              type:
                const: stt.error
        additionalProperties: false
      examples:
        - name: sessionStarted
          payload:
            type: stt.session.started
            event_id: stt_evt_example_1
            sequence: 1
            provider: deepgram
            model: flux-general-en
            origin: router
            input_config:
              audio_format: pcm_16000
              commit_strategy: auto
              turn_detection: semantic
              interim_results: false
              keyterms:
                - AllModels
              eager_end_threshold: 0.5
              commit_origin: provider
            capabilities:
              states:
                - started
                - end_candidate
                - resumed
                - end
              turn_scopes:
                - turn
              predictive: true
        - name: partialTranscript
          payload:
            type: stt.transcript.partial
            event_id: stt_evt_example_2
            sequence: 2
            provider: deepgram
            model: nova-3
            session_id: session_123
            turn_id: turn_1
            origin: provider
            text: Hello wor
        - name: eagerEnd
          payload:
            type: stt.signal
            event_id: stt_evt_example_5
            sequence: 5
            provider: deepgram
            model: flux-general-en
            turn_id: turn_1
            origin: provider
            provider_event: EagerEndOfTurn
            state: end_candidate
            turn_scope: turn
            basis:
              - semantic
            confidence: 0.82
            text: Could you explain
        - name: endpoint
          payload:
            type: stt.signal
            event_id: stt_evt_example_8
            sequence: 8
            provider: cartesia
            model: ink-whisper
            turn_id: turn_2
            origin: router
            state: end
            turn_scope: utterance
            basis:
              - vad
            audio_end_ms: 1840
    elTtsInit:
      name: elTtsInit
      title: Init / BOS frame
      summary: Optional first frame carrying voice and generation settings.
      description: To initialize the stream, send `text` as a single space together with at least one of `voice_settings`, `generation_config`, or `pronunciation_dictionary_locators`. A single space without any of these settings is treated as text to synthesize.
      payload:
        type: object
        required:
          - text
        anyOf:
          - required:
              - voice_settings
          - required:
              - generation_config
          - required:
              - pronunciation_dictionary_locators
        properties:
          text:
            const: " "
          voice_settings:
            type: object
            additionalProperties: true
            description: Voice settings such as `stability`, `similarity_boost`, and `speed`.
          generation_config:
            type: object
            additionalProperties: true
            description: "ElevenLabs-only generation tuning, e.g. `chunk_length_schedule: [50]`."
          pronunciation_dictionary_locators:
            type: array
            items:
              type: object
              additionalProperties: true
            description: ElevenLabs pronunciation dictionary references.
      examples:
        - name: init
          payload:
            text: " "
            voice_settings:
              stability: 0.5
            generation_config:
              chunk_length_schedule:
                - 50
    elTtsTextChunk:
      name: elTtsTextChunk
      title: Text chunk
      summary: Append text; optionally trigger or force generation.
      description: "Send non-empty `text` to append it to the stream. `try_trigger_generation: true` requests generation when possible, while `flush: true` generates all buffered text after this chunk. ElevenLabs, Deepgram, Grok, and Fish support both controls; MiniMax ignores them."
      payload:
        type: object
        required:
          - text
        not:
          allOf:
            - required:
                - text
              properties:
                text:
                  const: " "
            - anyOf:
                - required:
                    - voice_settings
                - required:
                    - generation_config
                - required:
                    - pronunciation_dictionary_locators
        properties:
          text:
            type: string
            minLength: 1
            description: Text to append (must be non-empty; empty text means flush/close).
          try_trigger_generation:
            type: boolean
            description: Hint the provider to start generating buffered text.
          flush:
            type: boolean
            description: Force generation of the buffered text after this chunk.
      examples:
        - name: appendText
          payload:
            text: "Hello "
            try_trigger_generation: true
        - name: appendAndFlush
          payload:
            text: world.
            flush: true
    elTtsFlush:
      name: elTtsFlush
      title: Bare flush frame
      summary: Flush the buffered text without closing the stream.
      description: 'Send `{"text": "", "flush": true}` to generate all buffered text while keeping the stream open. To append text and then flush it, send non-empty `text` with `flush: true`.'
      payload:
        type: object
        required:
          - text
          - flush
        properties:
          text:
            const: ""
          flush:
            const: true
      examples:
        - name: flush
          payload:
            text: ""
            flush: true
    elTtsClose:
      name: elTtsClose
      title: EOS / close frame
      summary: Finish the stream after the buffered text is synthesized.
      description: "Send an empty `text` without `flush: true` to synthesize any remaining text and close the stream. Omitting `flush` or setting it to `false` closes the stream."
      payload:
        type: object
        required:
          - text
        properties:
          text:
            const: ""
          flush:
            not:
              const: true
            description: Must not be the literal `true` (that makes the frame a bare flush).
      examples:
        - name: close
          payload:
            text: ""
    elTtsAudioChunk:
      name: elTtsAudioChunk
      title: Audio chunk
      summary: One frame per audio chunk, base64-encoded.
      description: Each frame contains base64 audio and may include provider-supplied character timing in `alignment` and `normalizedAlignment`.
      payload:
        type: object
        required:
          - audio
        properties:
          audio:
            type: string
            description: Base64-encoded audio bytes in the requested `output_format`.
          alignment:
            type: object
            description: Optional character timing for the original synthesized text, when supplied by the model.
            required:
              - chars
              - charStartTimesMs
              - charDurationsMs
            properties:
              chars:
                type: array
                items:
                  type: string
              charStartTimesMs:
                type: array
                items:
                  type: number
              charDurationsMs:
                type: array
                items:
                  type: number
          normalizedAlignment:
            type: object
            description: Optional character timing for normalized synthesized text, when supplied by the model.
            required:
              - chars
              - charStartTimesMs
              - charDurationsMs
            properties:
              chars:
                type: array
                items:
                  type: string
              charStartTimesMs:
                type: array
                items:
                  type: number
              charDurationsMs:
                type: array
                items:
                  type: number
      examples:
        - name: audioChunk
          payload:
            audio: bXAzLWJ5dGVz
        - name: audioChunkWithAlignment
          payload:
            audio: bXAzLWJ5dGVz
            alignment:
              chars:
                - H
                - i
              charStartTimesMs:
                - 0
                - 40
              charDurationsMs:
                - 40
                - 35
            normalizedAlignment:
              chars:
                - h
                - i
              charStartTimesMs:
                - 0
                - 40
              charDurationsMs:
                - 40
                - 35
    elTtsFinal:
      name: elTtsFinal
      title: Terminal frame
      summary: Marks successful completion of the audio stream.
      description: Emitted exactly once before a clean completion close. A clean upstream close (code `1000` or `1005`) synthesizes the terminal frame and is normalized to `1000 / tts_complete`; abnormal upstream closes propagate their original code and reason with no terminal frame.
      payload:
        type: object
        required:
          - audio
          - isFinal
          - normalizedAlignment
          - alignment
        properties:
          audio:
            const: null
          isFinal:
            const: true
          normalizedAlignment:
            const: null
          alignment:
            const: null
        additionalProperties: false
      examples:
        - name: streamComplete
          payload:
            audio: null
            isFinal: true
            normalizedAlignment: null
            alignment: null
    elTtsError:
      name: elTtsError
      title: Error frame
      summary: Provider or validation error in an ElevenLabs-compatible envelope.
      payload:
        type: object
        required:
          - audio
          - error
        properties:
          audio:
            const: null
            description: Always `null` on error frames.
          error:
            type: string
            description: Human-readable error message.
          code:
            type: string
            description: Machine-readable error code, when known.
      examples:
        - name: providerError
          payload:
            audio: null
            error: The provider session ended unexpectedly.
    elSttInputAudioChunk:
      name: elSttInputAudioChunk
      title: Input audio chunk
      summary: Base64 audio and/or an explicit commit.
      description: "Carries base64 input audio. With `commit: true` and empty (or absent) `audio_base_64`, the frame is a commit-only signal that closes the current utterance (the SDK's `conn.commit()`, honored with `commit_strategy=manual`); with audio and `commit: true`, the audio is processed first and the utterance is then committed."
      payload:
        type: object
        required:
          - message_type
        properties:
          message_type:
            const: input_audio_chunk
          audio_base_64:
            type: string
            description: Base64-encoded audio in the declared `audio_format`. May be empty on commit-only frames.
          sample_rate:
            type: number
            description: Optional per-frame sample-rate override.
          commit:
            type: boolean
            description: Commit the current utterance after this frame's audio.
      examples:
        - name: audio
          payload:
            message_type: input_audio_chunk
            audio_base_64: cGNtLWJ5dGVz
            sample_rate: 16000
        - name: commitOnly
          payload:
            message_type: input_audio_chunk
            audio_base_64: ""
            commit: true
    elSttSessionStarted:
      name: elSttSessionStarted
      title: Session started
      summary: Emitted when the transcription session is ready.
      payload:
        type: object
        required:
          - message_type
          - config
        properties:
          message_type:
            const: session_started
          session_id:
            type: string
            description: Provider session id, when available.
          config:
            type: object
            additionalProperties: true
            description: Provider echo of the effective session configuration (may be empty).
      examples:
        - name: sessionStarted
          payload:
            message_type: session_started
            session_id: sess_123
            config:
              sample_rate: 16000
    elSttPartialTranscript:
      name: elSttPartialTranscript
      title: Partial transcript
      summary: In-progress transcript for the current utterance.
      payload:
        type: object
        required:
          - message_type
          - text
        properties:
          message_type:
            const: partial_transcript
          text:
            type: string
      examples:
        - name: partial
          payload:
            message_type: partial_transcript
            text: hello wor
    elSttCommittedTranscript:
      name: elSttCommittedTranscript
      title: Committed transcript
      summary: Final transcript for a committed utterance.
      payload:
        type: object
        required:
          - message_type
          - text
        properties:
          message_type:
            const: committed_transcript
          text:
            type: string
          language_code:
            type: string
            description: Language tag for the committed segment, when the provider reports one.
      examples:
        - name: committed
          payload:
            message_type: committed_transcript
            text: hello world
            language_code: en
    elSttCommittedTranscriptWithTimestamps:
      name: elSttCommittedTranscriptWithTimestamps
      title: Committed transcript with timestamps
      summary: Final transcript with word-level timing (requires `include_timestamps`).
      payload:
        type: object
        required:
          - message_type
          - text
          - words
        properties:
          message_type:
            const: committed_transcript_with_timestamps
          text:
            type: string
          words:
            type: array
            items:
              $ref: "#/components/schemas/elSttRealtimeWord"
            description: Word/character entries with ElevenLabs realtime timing.
          language_code:
            type: string
            description: Language tag for the committed segment, when the provider reports one.
      examples:
        - name: committedWithWords
          payload:
            message_type: committed_transcript_with_timestamps
            text: hello world
            words:
              - text: hello
                start: 0
                end: 0.5
                type: word
              - text: world
                start: 0.5
                end: 1
                type: word
    elSttError:
      name: elSttError
      title: Error frame
      summary: Protocol or provider error; `message_type` carries the error code.
      description: Unlike the other frames, `message_type` is not a fixed constant — it carries the error code itself. Documented authentication codes include `authentication_failed` and `invalid_api_key`; other values may be provider-specific. Applications should not depend on undocumented codes.
      payload:
        type: object
        required:
          - message_type
          - message
        properties:
          message_type:
            type: string
            description: Error code.
          message:
            type: string
            description: Human-readable error message.
      examples:
        - name: authFailed
          payload:
            message_type: invalid_api_key
            message: Authentication failed.
    oaiSessionUpdate:
      name: oaiSessionUpdate
      title: session.update
      summary: Configure the transcription session (send before audio).
      description: GA session configuration. `session.provider` and `session.provider_options` are **AllModels extensions** for selecting the provider and passing provider-specific connection options. After audio has started, only lock-compatible updates are accepted (see the channel description for the `session_locked` rule).
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: session.update
          event_id:
            type: string
          session:
            $ref: "#/components/schemas/oaiSessionUpdatePayload"
      examples:
        - name: configureSession
          payload:
            type: session.update
            session:
              type: transcription
              provider: deepgram
              provider_options:
                encoding: linear16
              audio:
                input:
                  format:
                    type: audio/pcm
                    rate: 24000
                  transcription:
                    model: deepgram/nova-3
                    language: en
                  turn_detection: null
    oaiTranscriptionSessionUpdate:
      name: oaiTranscriptionSessionUpdate
      title: transcription_session.update
      summary: Transcription-session alias of session.update.
      description: Identical semantics to `session.update` (including the AllModels `provider` / `provider_options` extensions and the session lock); the server acknowledges with `transcription_session.updated`.
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: transcription_session.update
          event_id:
            type: string
          session:
            $ref: "#/components/schemas/oaiSessionUpdatePayload"
      examples:
        - name: configureTranscriptionSession
          payload:
            type: transcription_session.update
            session:
              input_audio_format: pcm16
              input_audio_transcription:
                model: openai/gpt-realtime-whisper
    oaiInputAudioBufferAppend:
      name: oaiInputAudioBufferAppend
      title: input_audio_buffer.append
      summary: Append base64 audio to the input buffer.
      description: The first append locks the session configuration (provider, model, audio format, commit strategy).
      payload:
        type: object
        required:
          - type
          - audio
        properties:
          type:
            const: input_audio_buffer.append
          event_id:
            type: string
          audio:
            type: string
            description: Base64-encoded audio in the session's declared input format (default `pcm_24000`).
      examples:
        - name: append
          payload:
            type: input_audio_buffer.append
            audio: cGNtLWJ5dGVz
    oaiInputAudioBufferCommit:
      name: oaiInputAudioBufferCommit
      title: input_audio_buffer.commit
      summary: Commit the buffered audio as one utterance.
      description: Acknowledged with `input_audio_buffer.committed` (carrying the new `item_id`), then submitted for transcription. Used with manual commit strategy (turn detection disabled).
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: input_audio_buffer.commit
          event_id:
            type: string
      examples:
        - name: commit
          payload:
            type: input_audio_buffer.commit
    oaiInputAudioBufferClear:
      name: oaiInputAudioBufferClear
      title: input_audio_buffer.clear
      summary: Drop the buffered, uncommitted audio.
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: input_audio_buffer.clear
          event_id:
            type: string
      examples:
        - name: clear
          payload:
            type: input_audio_buffer.clear
    oaiClose:
      name: oaiClose
      title: close
      summary: Finish and close the session.
      description: A non-standard convenience frame; closing the WebSocket works too.
      payload:
        type: object
        required:
          - type
        properties:
          type:
            const: close
      examples:
        - name: close
          payload:
            type: close
    oaiSessionCreated:
      name: oaiSessionCreated
      title: session.created
      summary: Emitted when the transcription session is ready.
      payload:
        type: object
        required:
          - type
          - event_id
          - session
        properties:
          type:
            const: session.created
          event_id:
            type: string
          session:
            $ref: "#/components/schemas/oaiSessionEcho"
      examples:
        - name: created
          payload:
            type: session.created
            event_id: event_1
            session:
              type: transcription
              audio:
                input:
                  format:
                    type: audio/pcm
                    rate: 24000
                  transcription:
                    model: deepgram/nova-3
    oaiSessionUpdated:
      name: oaiSessionUpdated
      title: session.updated
      summary: Acknowledges a session.update with the effective configuration.
      payload:
        type: object
        required:
          - type
          - event_id
          - session
        properties:
          type:
            const: session.updated
          event_id:
            type: string
          session:
            $ref: "#/components/schemas/oaiSessionEcho"
      examples:
        - name: updated
          payload:
            type: session.updated
            event_id: event_2
            session:
              type: transcription
              audio:
                input:
                  format:
                    type: audio/pcm
                    rate: 24000
                  transcription:
                    model: deepgram/nova-3
                    language: en
    oaiTranscriptionSessionUpdated:
      name: oaiTranscriptionSessionUpdated
      title: transcription_session.updated
      summary: Acknowledges a transcription_session.update with the effective configuration.
      payload:
        type: object
        required:
          - type
          - event_id
          - session
        properties:
          type:
            const: transcription_session.updated
          event_id:
            type: string
          session:
            $ref: "#/components/schemas/oaiSessionEcho"
      examples:
        - name: updated
          payload:
            type: transcription_session.updated
            event_id: event_2
            session:
              type: transcription
              audio:
                input:
                  format:
                    type: audio/pcm
                    rate: 24000
                  transcription:
                    model: openai/gpt-realtime-whisper
    oaiInputAudioBufferSpeechStarted:
      name: oaiInputAudioBufferSpeechStarted
      title: input_audio_buffer.speech_started
      summary: Reports that the selected provider detected speech onset.
      payload:
        type: object
        required:
          - type
          - event_id
          - audio_start_ms
          - item_id
        properties:
          type:
            const: input_audio_buffer.speech_started
          event_id:
            type: string
          audio_start_ms:
            type: integer
            description: Audio offset in ms where the provider detected speech onset; 0 when the provider reports no offset.
          item_id:
            type: string
            description: Id of the item this speech segment transcribes into; matches the eventual .committed/.completed.
      examples:
        - name: speech started
          payload:
            type: input_audio_buffer.speech_started
            event_id: event_3
            audio_start_ms: 120
            item_id: item_1
    oaiInputAudioBufferSpeechStopped:
      name: oaiInputAudioBufferSpeechStopped
      title: input_audio_buffer.speech_stopped
      summary: Reports that the selected provider detected speech end.
      payload:
        type: object
        required:
          - type
          - event_id
          - audio_end_ms
          - item_id
        properties:
          type:
            const: input_audio_buffer.speech_stopped
          event_id:
            type: string
          audio_end_ms:
            type: integer
            description: Audio offset in ms where the provider detected speech end; defaults to buffered audio duration when the provider reports no offset.
          item_id:
            type: string
            description: Id of the item this speech segment transcribes into; matches the eventual .committed/.completed.
      examples:
        - name: speech stopped
          payload:
            type: input_audio_buffer.speech_stopped
            event_id: event_4
            audio_end_ms: 980
            item_id: item_1
    oaiInputAudioBufferCommitted:
      name: oaiInputAudioBufferCommitted
      title: input_audio_buffer.committed
      summary: Acknowledges a commit and names the new transcript item.
      payload:
        type: object
        required:
          - type
          - event_id
          - item_id
        properties:
          type:
            const: input_audio_buffer.committed
          event_id:
            type: string
          item_id:
            type: string
            description: Id of the committed utterance (`item_1`, `item_2`, ...).
          previous_item_id:
            const: null
            description: Always `null` on this surface.
      examples:
        - name: committed
          payload:
            type: input_audio_buffer.committed
            event_id: event_3
            item_id: item_1
            previous_item_id: null
    oaiInputAudioBufferCleared:
      name: oaiInputAudioBufferCleared
      title: input_audio_buffer.cleared
      summary: Acknowledges an input_audio_buffer.clear.
      payload:
        type: object
        required:
          - type
          - event_id
        properties:
          type:
            const: input_audio_buffer.cleared
          event_id:
            type: string
      examples:
        - name: cleared
          payload:
            type: input_audio_buffer.cleared
            event_id: event_4
    oaiTranscriptionDelta:
      name: oaiTranscriptionDelta
      title: conversation.item.input_audio_transcription.delta
      summary: Incremental transcript text for the current item.
      payload:
        type: object
        required:
          - type
          - event_id
          - item_id
          - delta
        properties:
          type:
            const: conversation.item.input_audio_transcription.delta
          event_id:
            type: string
          item_id:
            type: string
          content_index:
            type: integer
            description: Always `0` on this surface.
          delta:
            type: string
            description: Appended transcript text.
      examples:
        - name: delta
          payload:
            type: conversation.item.input_audio_transcription.delta
            event_id: event_5
            item_id: item_1
            content_index: 0
            delta: "hello "
    oaiTranscriptionCompleted:
      name: oaiTranscriptionCompleted
      title: conversation.item.input_audio_transcription.completed
      summary: Final transcript and usage for a committed item.
      payload:
        type: object
        required:
          - type
          - event_id
          - item_id
          - transcript
          - usage
        properties:
          type:
            const: conversation.item.input_audio_transcription.completed
          event_id:
            type: string
          item_id:
            type: string
          content_index:
            type: integer
            description: Always `0` on this surface.
          transcript:
            type: string
            description: Full transcript for the committed utterance.
          usage:
            type: object
            required:
              - type
              - seconds
            properties:
              type:
                const: duration
              seconds:
                type: number
                description: Audio seconds attributed to the item.
      examples:
        - name: completed
          payload:
            type: conversation.item.input_audio_transcription.completed
            event_id: event_6
            item_id: item_1
            content_index: 0
            transcript: hello world
            usage:
              type: duration
              seconds: 1.28
    oaiError:
      name: oaiError
      title: error
      summary: OpenAI-shaped error envelope.
      description: "`error.code` carries the discriminating value — e.g. `session_locked` when a locked configuration change is rejected. Other values may be provider-specific. Applications should depend only on documented error codes. `error.param` is always `null`."
      payload:
        type: object
        required:
          - type
          - event_id
          - error
        properties:
          type:
            const: error
          event_id:
            type: string
          error:
            type: object
            required:
              - type
              - code
              - message
            properties:
              type:
                type: string
                description: OpenAI error family; AllModels-generated errors use `invalid_request_error`.
              code:
                type: string
                description: Machine-readable error code.
              message:
                type: string
              param:
                const: null
      examples:
        - name: sessionLocked
          payload:
            type: error
            event_id: event_7
            error:
              type: invalid_request_error
              code: session_locked
              message: Session configuration is locked after audio begins.
              param: null
    speechEngineAudioInput:
      name: speechEngineAudioInput
      title: User audio chunk
      summary: Raw binary audio or ElevenLabs-compatible base64 audio JSON.
      payload:
        oneOf:
          - type: string
            format: binary
          - type: object
            required:
              - user_audio_chunk
            properties:
              user_audio_chunk:
                type: string
                contentEncoding: base64
      examples:
        - name: base64Audio
          payload:
            user_audio_chunk: AAECAw==
    speechEnginePong:
      name: speechEnginePong
      title: Pong
      payload:
        type: object
        required:
          - type
          - event_id
        properties:
          type:
            const: pong
          event_id:
            type: integer
      examples:
        - name: pong
          payload:
            type: pong
            event_id: 1700000000
    speechEngineUserMessage:
      name: speechEngineUserMessage
      title: Typed user message
      payload:
        type: object
        required:
          - type
          - text
        properties:
          type:
            const: user_message
          text:
            type: string
      examples:
        - name: typedMessage
          payload:
            type: user_message
            text: I need help with my order.
    speechEngineMetadata:
      name: speechEngineMetadata
      title: Conversation initiation metadata
      payload:
        type: object
        required:
          - type
          - conversation_initiation_metadata_event
        properties:
          type:
            const: conversation_initiation_metadata
          conversation_initiation_metadata_event:
            type: object
            required:
              - conversation_id
              - agent_output_audio_format
              - user_input_audio_format
            properties:
              conversation_id:
                type: string
              agent_output_audio_format:
                type: string
              user_input_audio_format:
                type: string
      examples:
        - name: metadata
          payload:
            type: conversation_initiation_metadata
            conversation_initiation_metadata_event:
              conversation_id: conv_01
              agent_output_audio_format: pcm_16000
              user_input_audio_format: pcm_16000
    speechEngineUserTranscript:
      name: speechEngineUserTranscript
      title: User transcript
      payload:
        type: object
        required:
          - type
          - user_transcription_event
        properties:
          type:
            const: user_transcript
          user_transcription_event:
            type: object
            required:
              - event_id
              - user_transcript
            properties:
              event_id:
                type: integer
              user_transcript:
                type: string
      examples:
        - name: transcript
          payload:
            type: user_transcript
            user_transcription_event:
              event_id: 1
              user_transcript: I need help with my order.
    speechEngineAgentResponse:
      name: speechEngineAgentResponse
      title: Agent response text
      payload:
        type: object
        required:
          - type
          - agent_response_event
        properties:
          type:
            const: agent_response
          agent_response_event:
            type: object
            required:
              - event_id
              - agent_response
            properties:
              event_id:
                type: integer
              agent_response:
                type: string
      examples:
        - name: response
          payload:
            type: agent_response
            agent_response_event:
              event_id: 1
              agent_response: Certainly. What is your order number?
    speechEngineAudioOutput:
      name: speechEngineAudioOutput
      title: Synthesized agent audio
      payload:
        type: object
        required:
          - type
          - audio_event
        properties:
          type:
            const: audio
          audio_event:
            type: object
            required:
              - event_id
              - audio_base_64
            properties:
              event_id:
                type: integer
              audio_base_64:
                type: string
                contentEncoding: base64
      examples:
        - name: audio
          payload:
            type: audio
            audio_event:
              event_id: 1
              audio_base_64: AAECAw==
    speechEnginePing:
      name: speechEnginePing
      title: Ping
      payload:
        type: object
        required:
          - type
          - ping_event
        properties:
          type:
            const: ping
          ping_event:
            type: object
            required:
              - event_id
              - ping_ms
            properties:
              event_id:
                type: integer
              ping_ms:
                type: integer
      examples:
        - name: ping
          payload:
            type: ping
            ping_event:
              event_id: 1700000000
              ping_ms: 20000
    speechEngineInterruption:
      name: speechEngineInterruption
      title: Interruption
      payload:
        type: object
        required:
          - type
          - interruption_event
        properties:
          type:
            const: interruption
          interruption_event:
            type: object
            required:
              - event_id
            properties:
              event_id:
                type: integer
      examples:
        - name: interruption
          payload:
            type: interruption
            interruption_event:
              event_id: 1
    speechEngineResponseComplete:
      name: speechEngineResponseComplete
      title: Agent response complete
      payload:
        type: object
        required:
          - type
          - agent_response_complete_event
        properties:
          type:
            const: agent_response_complete
          agent_response_complete_event:
            type: object
            required:
              - event_id
            properties:
              event_id:
                type: integer
      examples:
        - name: complete
          payload:
            type: agent_response_complete
            agent_response_complete_event:
              event_id: 1
    speechEngineActivityStarted:
      name: speechEngineActivityStarted
      title: Speech activity started
      summary: "AllModels extension: the STT provider reported that the user started speaking."
      description: |
        Emitted when the selected STT model reports explicit speech-activity
        start (see the `/v1/stt` capability contract — models without activity
        signals never produce this frame). `event_id` is the identifier the
        next completed user turn will carry, so a client can attach barge-in
        state to a turn before its transcript exists.

        Unless the Speech Engine sets `conversation.interrupt_on` to
        `turn_committed`, this is also the point at which AllModels interrupts
        any response already playing.
      payload:
        type: object
        required:
          - type
          - event_id
        properties:
          type:
            const: allmodels_activity_started
          event_id:
            type: integer
            description: Identifier of the user turn this activity belongs to.
      examples:
        - name: activityStarted
          payload:
            type: allmodels_activity_started
            event_id: 2
    speechEngineActivityStopped:
      name: speechEngineActivityStopped
      title: Speech activity stopped
      summary: "AllModels extension: the STT provider reported that the user stopped speaking."
      description: |
        Counterpart of `allmodels_activity_started`. Activity stop is not a turn
        boundary: the completed transcript still arrives separately as
        `user_transcript`. Only models that advertise a `stopped` activity state
        emit this frame.
      payload:
        type: object
        required:
          - type
          - event_id
        properties:
          type:
            const: allmodels_activity_stopped
          event_id:
            type: integer
            description: Identifier of the user turn this activity belongs to.
      examples:
        - name: activityStopped
          payload:
            type: allmodels_activity_stopped
            event_id: 2
    speechEngineTranscriptDelta:
      name: speechEngineTranscriptDelta
      title: Interim user transcript
      summary: "AllModels extension: mutable in-progress transcript for the pending user turn."
      description: |
        Opt-in: sent only when the Speech Engine sets
        `conversation.upstream.interim_transcripts` to `true`. `text` is the full
        current hypothesis for the pending turn, not a delta to append, and it
        may be revised or withdrawn until `user_transcript` arrives.
      payload:
        type: object
        required:
          - type
          - text
          - event_id
        properties:
          type:
            const: allmodels_transcript_delta
          text:
            type: string
            description: Full mutable transcript hypothesis for the pending turn.
          event_id:
            type: integer
            description: Identifier the pending user turn will carry.
      examples:
        - name: interimTranscript
          payload:
            type: allmodels_transcript_delta
            text: I need help with my
            event_id: 2
    speechEngineAudioComplete:
      name: speechEngineAudioComplete
      title: Response audio complete
      summary: "AllModels extension: every audio frame for one response has been sent."
      description: |
        Emitted after the last `audio` frame of a response, when the TTS provider
        signalled end of stream. It is not sent for a response that was
        interrupted (an `interruption` frame is sent instead) or one whose TTS
        session failed.
      payload:
        type: object
        required:
          - type
          - event_id
        properties:
          type:
            const: allmodels_audio_complete
          event_id:
            type: integer
            description: Identifier of the completed response turn.
      examples:
        - name: audioComplete
          payload:
            type: allmodels_audio_complete
            event_id: 1
    speechEngineLatency:
      name: speechEngineLatency
      title: Stage latency
      summary: "AllModels extension: measured per-stage latency for one conversation turn."
      description: |
        Opt-in debug telemetry: sent only when the signed URL carries
        `allmodels_latency=true`. `stt` is provider time-to-first-transcript for
        the turn's audio, `llm` is the customer application's time to first
        response text (or the value the application self-reported with its own
        `allmodels_latency` frame), and `tts` is provider time-to-first-audio.
        Stages are emitted as they complete, so a turn may produce one, two, or
        three frames.
      payload:
        type: object
        required:
          - type
          - allmodels_latency_event
        properties:
          type:
            const: allmodels_latency
          allmodels_latency_event:
            type: object
            required:
              - event_id
              - stage
              - duration_ms
            properties:
              event_id:
                type: integer
                description: Conversation turn the measurement belongs to.
              stage:
                enum:
                  - stt
                  - llm
                  - tts
              duration_ms:
                type: integer
                minimum: 0
                description: Rounded stage duration in milliseconds.
      examples:
        - name: sttLatency
          payload:
            type: allmodels_latency
            allmodels_latency_event:
              event_id: 2
              stage: stt
              duration_ms: 180
    speechEngineClientError:
      name: speechEngineClientError
      title: Client-safe error
      summary: Session-fatal or per-turn failure in a client-safe envelope.
      description: |
        `error_event.code` is an HTTP-style status (`401` invalid key, `402`
        insufficient balance, `500` everything else) and `error_event.error_name`
        is the stable failure token.

        Two lifecycles share this frame:
        - **Session-fatal** — admission, grant, limit, pricing, provider, and
          application-timeout failures. The frame is sent first and the socket
          then closes with `4401` (401), `4402` (402), or `4403` (500). Tokens
          include `invalid_api_key`, `insufficient_balance`,
          `speech_engine_changed`, `grant_replayed`, `concurrency_limit`,
          `daily_limit`, `model_not_priced`, and `application_response_timeout`.
        - **Per-turn** — `application_response_failed`, emitted when the customer
          application reports an error for the turn in flight. The active
          response is interrupted, but the socket stays open and the
          conversation continues with the next user utterance.

        Applications should branch on `error_name` and treat an unknown token as
        fatal only when the socket also closes.
      payload:
        type: object
        required:
          - type
          - error_event
        properties:
          type:
            const: error
          error_event:
            type: object
            required:
              - code
              - error_name
              - message
            properties:
              code:
                type: integer
                description: "HTTP-style status: 401, 402, or 500."
              error_name:
                type: string
                description: Stable failure token.
              message:
                type: string
                description: Client-safe human-readable explanation.
      examples:
        - name: authFailure
          summary: "Session-fatal: the socket closes with 4401 immediately after this frame."
          payload:
            type: error
            error_event:
              code: 401
              error_name: invalid_api_key
              message: Invalid API key.
        - name: turnFailure
          summary: "Per-turn: the response is interrupted and the conversation stays open."
          payload:
            type: error
            error_event:
              code: 500
              error_name: application_response_failed
              message: Voice response failed.
    nativeRealtimeClientEvent:
      name: nativeRealtimeClientEvent
      title: AllModels Realtime client event
      payload:
        type: object
        required:
          - type
        properties:
          type:
            enum:
              - realtime.input_audio.append
              - realtime.input_audio.commit
              - realtime.input_audio.clear
              - realtime.input_text
              - realtime.response.cancel
              - realtime.pong
              - realtime.session.close
        additionalProperties: true
      examples:
        - name: appendBase64Audio
          payload:
            type: realtime.input_audio.append
            audio: AAECAw==
    nativeRealtimeServerEvent:
      name: nativeRealtimeServerEvent
      title: AllModels Realtime server event
      payload:
        type: object
        required:
          - type
          - event_id
          - sequence
          - session_id
        properties:
          type:
            pattern: ^realtime\.
          event_id:
            type: string
          sequence:
            type: integer
          session_id:
            type: string
        additionalProperties: true
      examples:
        - name: sessionCreated
          payload:
            type: realtime.session.created
            event_id: evt_01
            sequence: 1
            session_id: rts_01
            profile_id: rtp_01
  schemas:
    sttSignalFields:
      title: Normalized flat STT signal fields
      description: Fields used directly on `stt.signal`. The event has one lifecycle state rather than separate activity and boundary facets.
      type: object
      properties:
        state:
          enum:
            - started
            - end_candidate
            - resumed
            - end
          description: Flat provider signal state. `end_candidate` is predictive and may be followed by `resumed` or `end`.
        turn_scope:
          enum:
            - utterance
            - turn
          description: Acoustic utterance boundary or modeled conversational turn boundary. This is distinct from transcript-final `scope`.
        basis:
          type: array
          minItems: 1
          uniqueItems: true
          items:
            enum:
              - vad
              - silence
              - acoustic
              - linguistic
              - semantic
              - provider_declared
        audio_start_ms:
          type: number
          description: Provider audio-stream offset at signal start, in milliseconds.
        audio_end_ms:
          type: number
          description: Provider audio-stream offset at signal end, in milliseconds.
        confidence:
          type: number
          description: Provider-defined confidence; values are not calibrated across providers.
      additionalProperties: false
    sttSignalCapabilities:
      title: Active normalized signal capabilities
      description: Request-scoped capabilities after model selection and provider-option activation.
      type: object
      required:
        - states
        - turn_scopes
        - predictive
      properties:
        states:
          type: array
          uniqueItems: true
          items:
            enum:
              - started
              - end_candidate
              - resumed
              - end
          description: States this session may emit. An empty array means the session never emits `stt.signal`.
        turn_scopes:
          type: array
          uniqueItems: true
          items:
            enum:
              - utterance
              - turn
        predictive:
          type: boolean
          description: True if and only if `states` includes `end_candidate`.
      additionalProperties: false
    sttTranscriptWord:
      title: Transcribed word
      description: Word entry in an AllModels partial or committed transcript event.
      type: object
      required:
        - text
        - start
        - end
      properties:
        text:
          type: string
        start:
          type: number
          description: Word start, in seconds.
        end:
          type: number
          description: Word end, in seconds.
        type:
          type: string
          enum:
            - word
            - spacing
            - audio_event
        speaker_id:
          type: string
        logprob:
          type: number
        characters:
          type: array
          description: Per-character timestamps, when the provider exposes them.
          items:
            type: object
            required:
              - text
              - start
              - end
            properties:
              text:
                type: string
              start:
                type: number
              end:
                type: number
    elSttRealtimeWord:
      title: ElevenLabs realtime word
      description: Word/character entry in an ElevenLabs realtime committed timestamp frame. Unlike native transcript words, `type` is limited to `word` or `spacing` and `characters` is a list of character strings.
      type: object
      required:
        - text
        - start
        - end
      properties:
        text:
          type: string
        start:
          type: number
          description: Word start, in seconds.
        end:
          type: number
          description: Word end, in seconds.
        type:
          type: string
          enum:
            - word
            - spacing
        speaker_id:
          type: string
        logprob:
          type: number
        characters:
          type: array
          description: Per-character strings, when the provider exposes them.
          items:
            type: string
    oaiSessionUpdatePayload:
      title: OpenAI session update payload
      description: Use the GA nested `audio.input.*` fields to configure a session. `provider` and `provider_options` are AllModels extensions. The legacy flat fields remain accepted for compatibility.
      type: object
      properties:
        type:
          type: string
          description: Session type; this surface implements `transcription`.
        provider:
          type: string
          description: "AllModels extension: pin the STT provider (e.g. `deepgram`, `soniox`)."
        provider_options:
          type: object
          additionalProperties: true
          description: "AllModels extension: provider-specific connection options such as `encoding`, `sample_rate`, and `commit_strategy`."
        include:
          type: array
          items:
            type: string
          description: Extra event fields to include, forwarded as a provider option.
        input_audio_format:
          type: string
          description: Compatibility alias for `audio.input.format`; prefer the GA nested field.
        input_audio_transcription:
          $ref: "#/components/schemas/oaiTranscriptionConfig"
        audio:
          type: object
          properties:
            input:
              type: object
              properties:
                format:
                  description: GA format object (`{"type":"audio/pcm","rate":24000}`) or a flat string token.
                  oneOf:
                    - type: object
                      properties:
                        type:
                          type: string
                        rate:
                          type: integer
                      additionalProperties: true
                    - type: string
                transcription:
                  $ref: "#/components/schemas/oaiTranscriptionConfig"
                turn_detection:
                  description: Server turn detection. `null` (or type `none`) declares manual commits; any configured detector means server-driven (VAD) commits.
                noise_reduction:
                  description: Forwarded to the provider as an option, where supported.
      additionalProperties: true
    oaiTranscriptionConfig:
      title: OpenAI transcription config
      type: object
      properties:
        model:
          type: string
          description: Transcription model, preferably a `{author}/{modelName}` slug.
        language:
          type: string
        prompt:
          type: string
          description: Forwarded to the provider as an option, where supported.
        delay:
          type: string
          description: Forwarded to the provider as an option, where supported.
      additionalProperties: true
    oaiSessionEcho:
      title: OpenAI session echo
      description: The effective session configuration returned by AllModels.
      type: object
      required:
        - type
        - audio
      properties:
        type:
          const: transcription
        audio:
          type: object
          required:
            - input
          properties:
            input:
              type: object
              required:
                - format
                - transcription
              properties:
                format:
                  type: object
                  required:
                    - type
                    - rate
                  properties:
                    type:
                      const: audio/pcm
                    rate:
                      type: integer
                      description: Effective input sample rate (default 24000).
                transcription:
                  type: object
                  required:
                    - model
                  properties:
                    model:
                      type: string
                    language:
                      type: string
                    prompt:
                      type: string
                    delay:
                      type: string
                turn_detection:
                  description: Echo of the session's turn-detection value, when set.
