{
  "openapi": "3.1.0",
  "info": {
    "title": "EdgeSpeak Local API",
    "version": "v1",
    "summary": "Local-first transcription, alignment, speech and language APIs served from your own machine.",
    "description": "EdgeSpeak runs transcription, alignment, speech synthesis and language models on your own hardware and exposes them over an OpenAI-compatible HTTP API. There are two ways to serve it: the desktop app gateway on `http://127.0.0.1:1117/v1`, and the headless service started with `edgespeak-cli serve` on `http://127.0.0.1:1118/v1`. Both share the same gateway implementation, so paths and request shapes are identical; only the base URL and the API key differ. Two inputs are desktop-only: streaming transcription (`stream=true`) and server-local file paths (`text_path` on alignments, `file` on segmentations) are refused by the headless listener.\n\nAuthenticate with `Authorization: Bearer <API key>` or `x-api-key: <API key>`. The API Key authenticates requests and is separate from the License Key that activates the software. Local authentication is off on fresh desktop installs, and the headless service enables it when `EDGESPEAK_API_KEY` is set to a non-empty value at startup. The health aliases are public and need no key.\n\nThe general request-body limit is 512 MiB. A healthy service is not the same as a loaded model: inspect `GET /v1/models`, then download and load what you need before inference.\n\nThe Realtime WebSocket endpoint `WS /v1/realtime` is not described here because OpenAPI does not model WebSocket sessions; see https://edgespeak.com/docs/api-realtime for its events and connection example."
  },
  "externalDocs": {
    "description": "EdgeSpeak Local Gateway API documentation",
    "url": "https://edgespeak.com/docs/api"
  },
  "servers": [
    {
      "url": "http://127.0.0.1:1117/v1",
      "description": "Desktop app gateway"
    },
    {
      "url": "http://127.0.0.1:1118/v1",
      "description": "Headless service started with edgespeak-cli serve"
    }
  ],
  "tags": [
    {
      "name": "Models and service",
      "description": "Service health, the model catalog, and the download, load and unload lifecycle.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/api-models"
      }
    },
    {
      "name": "Text processing",
      "description": "Sentence and paragraph segmentation, and written-to-spoken text normalization.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/api-text"
      }
    },
    {
      "name": "Transcription and speakers",
      "description": "Transcription, speaker timelines, embeddings and similarity.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/api-audio"
      }
    },
    {
      "name": "Forced alignment",
      "description": "Align known text with audio for word and segment timestamps, with progress and concurrency handling.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/api-align"
      }
    },
    {
      "name": "Speech and voice library",
      "description": "Speech synthesis and the reusable local voice library.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/api-speech"
      }
    },
    {
      "name": "Language and vision",
      "description": "OpenAI Responses and Chat Completions, Anthropic Messages, tokenization and multimodal input.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/api-language"
      }
    },
    {
      "name": "MCP tools",
      "description": "Streamable HTTP transport for the local MCP tools.",
      "externalDocs": {
        "url": "https://edgespeak.com/docs/mcp"
      }
    }
  ],
  "security": [
    {
      "bearerAuth": []
    },
    {
      "apiKeyAuth": []
    }
  ],
  "paths": {
    "/health": {
      "get": {
        "operationId": "checkHealth",
        "tags": [
          "Models and service"
        ],
        "summary": "Check service health",
        "description": "Public health alias at the server root, outside `/v1`. It requires no API key. A running service can return HTTP 200 with an inactive license, and a healthy service does not prove that any model is loaded. See https://edgespeak.com/docs/api-models#health.",
        "servers": [
          {
            "url": "http://127.0.0.1:1117",
            "description": "Desktop app gateway"
          },
          {
            "url": "http://127.0.0.1:1118",
            "description": "Headless service started with edgespeak-cli serve"
          }
        ],
        "security": [],
        "responses": {
          "200": {
            "description": "The service is reachable. `license` is `active` or `inactive`; check both fields.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/HealthStatus"
                },
                "example": {
                  "status": "ok",
                  "license": "active"
                }
              }
            }
          }
        }
      }
    },
    "/v1/health": {
      "get": {
        "operationId": "checkHealthV1",
        "tags": [
          "Models and service"
        ],
        "summary": "Check service health",
        "description": "Versioned alias of the root health check, with the same public, unauthenticated behavior and the same response body. See https://edgespeak.com/docs/api-models#health.",
        "servers": [
          {
            "url": "http://127.0.0.1:1117",
            "description": "Desktop app gateway"
          },
          {
            "url": "http://127.0.0.1:1118",
            "description": "Headless service started with edgespeak-cli serve"
          }
        ],
        "security": [],
        "responses": {
          "200": {
            "description": "The service is reachable. `license` is `active` or `inactive`; check both fields.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/HealthStatus"
                },
                "example": {
                  "status": "ok",
                  "license": "active"
                }
              }
            }
          }
        }
      }
    },
    "/models": {
      "get": {
        "operationId": "listModels",
        "tags": [
          "Models and service"
        ],
        "summary": "Discover models and capabilities",
        "description": "Lists the models known to this service, including entries that are not downloaded or loaded. Filter `supported_endpoints` for the API you want, read `default_for` for defaults, and check `execution_location` before sending content. Alignment models list `/v1/audio/alignments`: `EdgeSpeak/Lattice-2` is the default and covers every language in the catalog, mixed-language text and `und`, while `EdgeSpeak/Lattice-1` covers Chinese, English and German only and must be requested explicitly. See https://edgespeak.com/docs/api-models#catalog.",
        "responses": {
          "200": {
            "description": "The current catalog. The example is a single-entry excerpt; a real catalog differs.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/ModelList"
                },
                "example": {
                  "object": "list",
                  "data": [
                    {
                      "id": "Qwen/Qwen3.5-4B",
                      "object": "model",
                      "created": 0,
                      "owned_by": "EdgeSpeak",
                      "supported_endpoints": [
                        "/v1/chat/completions",
                        "/v1/responses",
                        "/v1/messages",
                        "/v1/tokenize"
                      ],
                      "features": [
                        "reasoning",
                        "vision",
                        "tool_calling"
                      ],
                      "execution_location": "local",
                      "default_for": []
                    }
                  ]
                }
              }
            }
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/models/download": {
      "post": {
        "operationId": "startModelDownload",
        "tags": [
          "Models and service"
        ],
        "summary": "Start a model download",
        "description": "Starts a background download job on the service host and returns 202 Accepted, not a ready model. Downloads consume network bandwidth and disk space. Built-in `EdgeSpeak/Skylark` is already available and has no download, load or unload lifecycle. Poll `GET /v1/models/downloads` before loading. See https://edgespeak.com/docs/api-models#download.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/ModelLifecycleRequest"
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B"
              }
            }
          }
        },
        "responses": {
          "202": {
            "description": "The download job was accepted. `total_bytes: 0` can mean the total is not known yet.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/DownloadJob"
                },
                "example": {
                  "model": "Qwen/Qwen3.5-4B",
                  "state": "downloading",
                  "bytes_downloaded": 0,
                  "total_bytes": 0,
                  "error": null
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/models/downloads": {
      "get": {
        "operationId": "listModelDownloads",
        "tags": [
          "Models and service"
        ],
        "summary": "Check download progress",
        "description": "Polls download jobs, for example once per second. Match the entry by `model`; an empty list does not prove installation. On failure inspect `error`, and retry with an explicit new POST. See https://edgespeak.com/docs/api-models#downloads.",
        "responses": {
          "200": {
            "description": "Current download jobs. Byte counts in the example are illustrative.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/DownloadJobList"
                },
                "example": {
                  "object": "list",
                  "data": [
                    {
                      "model": "Qwen/Qwen3.5-4B",
                      "state": "completed",
                      "bytes_downloaded": 2500000000,
                      "total_bytes": 2500000000,
                      "error": null
                    }
                  ]
                }
              }
            }
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/models/download/cancel": {
      "post": {
        "operationId": "cancelModelDownload",
        "tags": [
          "Models and service"
        ],
        "summary": "Cancel an active download",
        "description": "Cancels a currently active download job. Cancellation is asynchronous; poll `GET /v1/models/downloads` until the job reaches a terminal state. See https://edgespeak.com/docs/api-models#cancel-download.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/ModelLifecycleRequest"
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B"
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The cancellation was accepted.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/DownloadCancelResult"
                },
                "example": {
                  "model": "Qwen/Qwen3.5-4B",
                  "state": "cancelling"
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/models/load": {
      "post": {
        "operationId": "loadModel",
        "tags": [
          "Models and service"
        ],
        "summary": "Load a local model",
        "description": "Loads a downloaded model and allocates memory on the service host. Local language models must be loaded before language generation or Realtime use. See https://edgespeak.com/docs/api-models#load.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/ModelLifecycleRequest"
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B"
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The model is loaded. Host-specific fields may also appear.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/ModelLifecycleResult"
                },
                "example": {
                  "success": true,
                  "model": "Qwen/Qwen3.5-4B",
                  "status": "loaded"
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/models/running": {
      "get": {
        "operationId": "listRunningModels",
        "tags": [
          "Models and service"
        ],
        "summary": "Inspect running models",
        "description": "Reports runtime state rather than catalog presence. During a cold load or a model switch `runtime.state` can be `loading` with a temporarily empty model list, which is not completed unloading. Builds containing the queue-status change also return `runtime.queues`. See https://edgespeak.com/docs/api-models#running.",
        "responses": {
          "200": {
            "description": "Runtime state. The contents of each model's `status` are model-specific.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/RunningModels"
                },
                "example": {
                  "object": "list",
                  "data": [
                    {
                      "id": "Qwen/Qwen3.5-4B",
                      "status": {
                        "value": "loaded"
                      }
                    }
                  ],
                  "runtime": {
                    "state": "running"
                  }
                }
              }
            }
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/models/unload": {
      "post": {
        "operationId": "unloadModel",
        "tags": [
          "Models and service"
        ],
        "summary": "Unload a model",
        "description": "Releases a model from memory without deleting its downloaded files. Active sessions can prevent unloading; finish them and retry. See https://edgespeak.com/docs/api-models#unload.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/ModelLifecycleRequest"
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B"
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The model is unloaded.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/ModelLifecycleResult"
                },
                "example": {
                  "success": true,
                  "model": "Qwen/Qwen3.5-4B",
                  "status": "unloaded"
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/text/segmentations": {
      "post": {
        "operationId": "segmentText",
        "tags": [
          "Text processing"
        ],
        "summary": "Segment sentences",
        "description": "Semantic sentence segmentation over a JSON body. Supply plain `text` or timed `segments[]`; the desktop gateway also accepts `file`, a server-local exported Transcript JSON path, while Headless requires inline content. Setting `paragraph_threshold` switches to paragraph mode, where each returned item is a paragraph and the top-level `text` joins paragraphs with a blank line. This endpoint always segments and does not accept `semantic_sentence_enabled`. See https://edgespeak.com/docs/api-text#sentences and https://edgespeak.com/docs/api#segmentation.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "properties": {
                  "text": {
                    "type": "string",
                    "description": "Plain text input. Paragraph mode requires this form."
                  },
                  "segments": {
                    "type": "array",
                    "items": {
                      "$ref": "#/components/schemas/Segment"
                    },
                    "description": "Timed input; timestamps and speakers can be preserved."
                  },
                  "file": {
                    "type": "string",
                    "description": "Desktop gateway only: absolute path to an exported Transcript JSON on the service host. Headless rejects it."
                  },
                  "threshold": {
                    "type": "number",
                    "minimum": 0,
                    "maximum": 1,
                    "default": 0.35,
                    "description": "Sentence threshold."
                  },
                  "paragraph_threshold": {
                    "type": "number",
                    "minimum": 0,
                    "maximum": 1,
                    "description": "Enables native model paragraph boundaries. Must be at least `threshold`. Omit for sentence mode."
                  },
                  "min_chars": {
                    "type": "integer",
                    "description": "Default length 12. Length constraints are off by default; supplying a length field enables them."
                  },
                  "max_chars": {
                    "type": "integer",
                    "description": "Default length 42."
                  },
                  "start_margin": {
                    "type": "number",
                    "description": "Seconds; default margin 0.2."
                  },
                  "end_margin": {
                    "type": "number",
                    "description": "Seconds; default margin 0.2."
                  },
                  "options": {
                    "type": "object",
                    "description": "`paragraph_threshold` and the length or margin fields may also be supplied here; a matching top-level field takes precedence."
                  }
                }
              },
              "example": {
                "text": "Hello world. Let us begin.",
                "threshold": 0.35
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "Segmented text. Plain text input has no timestamps or speakers.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/SegmentationResult"
                },
                "example": {
                  "task": "segment",
                  "text": "Hello world. Let us begin.",
                  "segments": [
                    {
                      "text": "Hello world."
                    },
                    {
                      "text": "Let us begin."
                    }
                  ]
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/text/normalizations": {
      "post": {
        "operationId": "normalizeText",
        "tags": [
          "Text processing"
        ],
        "summary": "Normalize text: TN and ITN",
        "description": "Converts written forms to spoken forms (`mode:\"tn\"`) or spoken forms to written forms (`mode:\"itn\"`) without audio. Unknown request fields are rejected, and this endpoint does not consume audio-duration quota. Supported languages and categories are in the Skylark catalog entry under `x_edgespeak.normalization`. See https://edgespeak.com/docs/api-text#normalize and https://edgespeak.com/docs/api#normalization.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": [
                  "text",
                  "language"
                ],
                "additionalProperties": false,
                "properties": {
                  "text": {
                    "type": "string",
                    "minLength": 1,
                    "description": "Required, non-empty, up to 256 KiB UTF-8."
                  },
                  "language": {
                    "type": "string",
                    "description": "Required: a language code, or `und` to choose readings segment by segment from the text. An unrecognized tag returns 400 `normalization_invalid_request`; a language without normalization rules returns 422 `normalization_language_unsupported`. See https://edgespeak.com/docs/languages#normalization for supported languages."
                  },
                  "model": {
                    "type": "string",
                    "default": "EdgeSpeak/Skylark",
                    "description": "Normalization model. Defaults to `EdgeSpeak/Skylark`."
                  },
                  "mode": {
                    "type": "string",
                    "enum": [
                      "tn",
                      "itn"
                    ],
                    "default": "tn",
                    "description": "`tn` rewrites written forms as spoken forms; `itn` does the reverse. Defaults to `tn`."
                  },
                  "top_k": {
                    "type": "integer",
                    "minimum": 1,
                    "maximum": 16,
                    "default": 8,
                    "description": "Number of alternatives to return, range 1-16. Defaults to 8; send `1` for the best candidate only."
                  },
                  "classes": {
                    "type": "array",
                    "minItems": 1,
                    "items": {
                      "type": "string"
                    },
                    "description": "Optional non-empty array of categories, for example `cardinal`, `date` or `money`."
                  }
                }
              },
              "example": {
                "text": "Registration is open for the workshop. The fee is $25 and the session starts at 09:30.",
                "language": "en",
                "mode": "tn",
                "top_k": 1
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "Normalization candidates. Span offsets are UTF-8 byte ranges in the original input.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/NormalizationResult"
                },
                "example": {
                  "task": "normalize",
                  "model": "EdgeSpeak/Skylark",
                  "language": "en",
                  "mode": "tn",
                  "text": "Registration is open for the workshop. The fee is $25 and the session starts at 09:30.",
                  "alternatives": [
                    {
                      "rank": 0,
                      "text": "Registration is open for the workshop. The fee is twenty five dollars and the session starts at nine thirty.",
                      "spans": [
                        {
                          "start_byte": 50,
                          "end_byte": 53,
                          "class": "money",
                          "source": "$25",
                          "output": "twenty five dollars"
                        },
                        {
                          "start_byte": 80,
                          "end_byte": 85,
                          "class": "time",
                          "source": "09:30",
                          "output": "nine thirty"
                        }
                      ]
                    }
                  ]
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/audio/transcriptions": {
      "post": {
        "operationId": "createTranscription",
        "tags": [
          "Transcription and speakers"
        ],
        "summary": "Transcribe an audio file",
        "description": "Transcribes an uploaded audio or video file. `json` returns text with duration, `verbose_json` returns a structured transcript, `text` returns plain text, and `diarized_json` also runs speaker diarization. Word or segment granularities require verbose or diarized output. `stream=true` is desktop-gateway only and returns SSE text deltas with JSON output and no granularities; the headless listener refuses it with 400. See https://edgespeak.com/docs/api-audio#transcribe and https://edgespeak.com/docs/api#audio-parameters.",
        "requestBody": {
          "required": true,
          "content": {
            "multipart/form-data": {
              "schema": {
                "type": "object",
                "required": [
                  "file"
                ],
                "properties": {
                  "file": {
                    "type": "string",
                    "format": "binary",
                    "description": "Audio or video file. The general request-body limit is 512 MiB."
                  },
                  "model": {
                    "type": "string",
                    "description": "Omit to use the service's configured default."
                  },
                  "language": {
                    "type": "string",
                    "description": "Optional; omit it to detect the language from the audio. Validated against the supported transcription languages. A code outside that set returns 400 `asr_language_unsupported` with `param=language`. See https://edgespeak.com/docs/languages#transcription for supported languages."
                  },
                  "response_format": {
                    "type": "string",
                    "enum": [
                      "json",
                      "verbose_json",
                      "text",
                      "diarized_json"
                    ],
                    "description": "Output mode."
                  },
                  "stream": {
                    "type": "boolean",
                    "description": "Desktop gateway only, with JSON output and no timestamp granularities."
                  },
                  "timestamp_granularities[]": {
                    "type": "array",
                    "items": {
                      "type": "string",
                      "enum": [
                        "word",
                        "segment"
                      ]
                    },
                    "description": "Requires verbose or diarized output."
                  },
                  "word_timestamp_enabled": {
                    "type": "boolean",
                    "description": "Controls word timestamps directly."
                  },
                  "candidates": {
                    "type": "integer",
                    "minimum": 1,
                    "maximum": 8,
                    "description": "N-best output. `1` is normal; `2`-`8` require non-streaming `verbose_json` without word timestamps. A conflicting repeated value returns 400."
                  },
                  "semantic_sentence_enabled": {
                    "type": "boolean",
                    "description": "Controls semantic processing. An explicit `false` conflicts with length or margin fields and returns 400."
                  },
                  "min_chars": {
                    "type": "integer",
                    "description": "Default length 12. Length constraints are off by default; supplying a length field enables them and requests semantic processing."
                  },
                  "max_chars": {
                    "type": "integer",
                    "description": "Default length 42."
                  },
                  "start_margin": {
                    "type": "number",
                    "description": "Seconds; default margin 0.2."
                  },
                  "end_margin": {
                    "type": "number",
                    "description": "Seconds; default margin 0.2."
                  }
                }
              },
              "encoding": {
                "file": {
                  "contentType": "application/octet-stream"
                },
                "timestamp_granularities[]": {
                  "style": "form",
                  "explode": true
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The transcript, in the shape selected by `response_format`. A streaming request returns SSE `transcript.text.delta` events followed by `transcript.text.done`; an HTTP 200 does not by itself prove that a stream completed.",
            "content": {
              "application/json": {
                "schema": {
                  "oneOf": [
                    {
                      "$ref": "#/components/schemas/TranscriptJson"
                    },
                    {
                      "$ref": "#/components/schemas/TranscriptVerbose"
                    },
                    {
                      "$ref": "#/components/schemas/TranscriptDiarized"
                    }
                  ]
                },
                "example": {
                  "text": "Hello world.",
                  "start": 0.0,
                  "duration": 2.5,
                  "language": "en",
                  "usage": {
                    "type": "duration",
                    "seconds": 2.5
                  }
                }
              },
              "text/plain": {
                "schema": {
                  "type": "string"
                },
                "example": "Hello world."
              },
              "text/event-stream": {
                "schema": {
                  "type": "string"
                },
                "example": "event: transcript.text.delta\ndata: {\"type\":\"transcript.text.delta\",\"delta\":\"Hello world.\"}\n\nevent: transcript.text.done\ndata: {\"type\":\"transcript.text.done\",\"text\":\"Hello world.\",\"usage\":{\"type\":\"duration\",\"seconds\":2.5}}\n"
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/audio/alignments": {
      "post": {
        "operationId": "createAlignment",
        "tags": [
          "Forced alignment"
        ],
        "summary": "Align known text with audio",
        "description": "Locates known words in audio rather than recognizing unknown speech. Send `file` and matching `text`; the desktop gateway also accepts an absolute local `text_path`, while Headless requires `text`. Timestamps stay relative to the original audio: `duration` is the full media duration and `usage.seconds` measures the processed window. Sending the `progress: 1` header returns an NDJSON progress stream whose result line carries `task: \"align\"`; parse each line independently, not as SSE. See https://edgespeak.com/docs/api-align#align and https://edgespeak.com/docs/api#alignment-options. A failed alignment returns 422 with an `error.alignment` object instead of a result; `search_effort` chooses the search range and can retry once with a wider one. See https://edgespeak.com/docs/api-align#alignment-failure.",
        "parameters": [
          {
            "name": "progress",
            "in": "header",
            "required": false,
            "schema": {
              "type": "string",
              "enum": [
                "1"
              ]
            },
            "description": "Set to `1` for an NDJSON alignment progress stream. Ordinary requests return one JSON response."
          }
        ],
        "requestBody": {
          "required": true,
          "content": {
            "multipart/form-data": {
              "schema": {
                "type": "object",
                "required": [
                  "file"
                ],
                "properties": {
                  "file": {
                    "type": "string",
                    "format": "binary",
                    "description": "Audio file."
                  },
                  "text": {
                    "type": "string",
                    "description": "Transcript to align. Required on Headless."
                  },
                  "text_path": {
                    "type": "string",
                    "description": "Desktop gateway only: absolute local path to the transcript."
                  },
                  "model": {
                    "type": "string",
                    "description": "Optional alignment model. The default `EdgeSpeak/Lattice-2` covers every language in the catalog, mixed-language text and `und`. `EdgeSpeak/Lattice-1` covers Chinese, English and German only and must be requested explicitly. A transcription or speech model returns 400 `model_capability_unsupported`, and an unknown ID returns 404 `model_not_found`; both carry `param: \"model\"`."
                  },
                  "language": {
                    "type": "string",
                    "description": "Optional hint. Omit it to let the runtime detect the language, send a code such as `zh` or `yue` to pin one, or send `und` to skip detection and assume no language. `EdgeSpeak/Lattice-1` rejects `und` and out-of-list codes with 400 `bad_request`, and a language it does not cover with 422 `language_unsupported`. See https://edgespeak.com/docs/languages#alignment for supported languages."
                  },
                  "start": {
                    "type": "number",
                    "description": "Window start in seconds. Supply together with `end`, with `0 <= start < end` and within the audio."
                  },
                  "end": {
                    "type": "number",
                    "description": "Window end in seconds."
                  },
                  "search_effort": {
                    "type": "string",
                    "enum": [
                      "standard",
                      "extended",
                      "auto",
                      ""
                    ],
                    "default": "standard",
                    "description": "Search range. `standard` searches once with the standard range; an empty value (`\"\"`) means the same. `extended` searches once with a wider range, slower and using more memory. `auto` runs `standard` and, only if it fails with `error.alignment.retry_recommended: true`, runs `extended` once. Other values return 400 `bad_request` with `param: \"search_effort\"`."
                  },
                  "protected_terms[]": {
                    "type": "array",
                    "items": {
                      "type": "string"
                    },
                    "description": "Terms that must align as a single word rather than being split, such as product names or jargon. Repeat the field, or send a JSON array in `protected_terms`."
                  },
                  "text_normalization": {
                    "type": "string",
                    "description": "A JSON string inside multipart. Omitting the whole field enables normalization; `{\"enabled\":false}` or an empty object disables it. Fields are `enabled`, `top_k` (default 8, range 1-16), `classes`, `apply_dictionary` (default false), `dictionary_ids` and `pins` with `{source_start_byte, source_end_byte, rank}` over the original UTF-8 transcript, where `rank: null` preserves the source. When disabled, do not supply `top_k`, `classes` or non-empty `pins`."
                  },
                  "audio_track": {
                    "type": "integer",
                    "description": "Zero-based decodable audio track. A conflicting repeated value returns 400."
                  },
                  "audio_channel": {
                    "type": "string",
                    "description": "`mix` or a zero-based channel index. A conflicting repeated value returns 400."
                  },
                  "semantic_sentence_enabled": {
                    "type": "boolean",
                    "description": "Controls semantic processing. An explicit `false` conflicts with length or margin fields and returns 400."
                  },
                  "min_chars": {
                    "type": "integer",
                    "description": "Default length 12. Length constraints are off by default; supplying a length field enables them and requests semantic processing."
                  },
                  "max_chars": {
                    "type": "integer",
                    "description": "Default length 42."
                  },
                  "start_margin": {
                    "type": "number",
                    "description": "Seconds; default margin 0.2."
                  },
                  "end_margin": {
                    "type": "number",
                    "description": "Seconds; default margin 0.2."
                  }
                }
              },
              "encoding": {
                "file": {
                  "contentType": "application/octet-stream"
                },
                "protected_terms[]": {
                  "style": "form",
                  "explode": true
                },
                "text_normalization": {
                  "contentType": "text/plain"
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The alignment result. With `progress: 1` the body is NDJSON: progress lines, a `{\"stage\":\"extended_search\"}` line when `search_effort=auto` starts the wider search (progress then restarts from 0), and a last line that is either the result (`task: \"align\"`) or the same `{\"error\": {...}}` envelope as the 422 body. The HTTP status of the stream stays 200; progress 1.0 alone does not prove success, and a stream that closes without a result or error line is a failure.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/AlignmentResult"
                },
                "example": {
                  "task": "align",
                  "duration": 2.5,
                  "text": "Hello world.",
                  "segments": [
                    {
                      "id": 0,
                      "start": 0.2,
                      "end": 1.4,
                      "text": "Hello world.",
                      "words": [
                        {
                          "word": "Hello",
                          "start": 0.2,
                          "end": 0.6,
                          "score": 0.98
                        },
                        {
                          "word": "world.",
                          "start": 0.7,
                          "end": 1.4,
                          "score": 0.96
                        }
                      ]
                    }
                  ],
                  "usage": {
                    "type": "duration",
                    "seconds": 2.5
                  }
                }
              },
              "application/x-ndjson": {
                "schema": {
                  "type": "string"
                },
                "example": "{\"progress\": 0.5}\n{\"progress\": 1.0}\n{\"task\":\"align\",\"duration\":2.5,\"text\":\"Hello world.\",\"segments\":[],\"usage\":{\"type\":\"duration\",\"seconds\":2.5}}\n"
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "404": {
            "$ref": "#/components/responses/404"
          },
          "413": {
            "$ref": "#/components/responses/413"
          },
          "422": {
            "description": "`alignment_failed`: alignment ran but could not be completed. `alignment_search_budget_exceeded`: a wider search would need more memory than this computer can spare, so it did not run. Both carry `error.alignment` with `reason` (`no_path`, `no_words`, `collapsed`, `search_incomplete`), `search_effort_used`, `search_path` (`whole_audio`, `streaming`), `search_limit` (`none`, `budget`, `graph_size`, `estimate_overflow`, `streaming`), `retry_recommended` (always present) and `estimated_extra_bytes` (an estimate of the extra memory a wider search needs, not a cap); unknown fields are omitted. Failed alignments are not counted in usage. The same status also covers `language_unsupported` (change the model, not the language code) and `normalization_class_unsupported`.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/Error"
                },
                "example": {
                  "error": {
                    "message": "Alignment could not be completed; the search stopped before the end of the text.",
                    "type": "invalid_request_error",
                    "code": "alignment_failed",
                    "alignment": {
                      "reason": "search_incomplete",
                      "search_effort_used": "standard",
                      "search_path": "whole_audio",
                      "search_limit": "none",
                      "retry_recommended": true,
                      "estimated_extra_bytes": 3221225472
                    }
                  }
                }
              }
            }
          }
        }
      }
    },
    "/speaker/diarizations": {
      "post": {
        "operationId": "createDiarization",
        "tags": [
          "Transcription and speakers"
        ],
        "summary": "Get a speaker activity timeline",
        "description": "Returns who spoke when, without transcription. Speaker endpoints accept uploaded bytes rather than arbitrary local paths, and `num_speakers` is clustering guidance, not an identity claim. See https://edgespeak.com/docs/api-audio#diarize.",
        "requestBody": {
          "required": true,
          "content": {
            "multipart/form-data": {
              "schema": {
                "type": "object",
                "required": [
                  "file"
                ],
                "properties": {
                  "file": {
                    "type": "string",
                    "format": "binary",
                    "description": "Audio or video file. The general request-body limit is 512 MiB."
                  },
                  "num_speakers": {
                    "type": "integer",
                    "minimum": 1,
                    "maximum": 32,
                    "description": "Optional clustering guidance."
                  }
                }
              },
              "encoding": {
                "file": {
                  "contentType": "application/octet-stream"
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The speaker timeline. No text is returned.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/DiarizationResult"
                },
                "example": {
                  "duration": 10.0,
                  "speakers": [
                    "SPEAKER_00",
                    "SPEAKER_01"
                  ],
                  "segments": [
                    {
                      "start": 0.4,
                      "end": 3.2,
                      "speaker": "SPEAKER_00"
                    },
                    {
                      "start": 4.0,
                      "end": 8.1,
                      "speaker": "SPEAKER_01"
                    }
                  ]
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/speaker/embeddings": {
      "post": {
        "operationId": "createSpeakerEmbeddings",
        "tags": [
          "Transcription and speakers"
        ],
        "summary": "Extract speaker embeddings",
        "description": "Extracts one 256-dimensional embedding per uploaded sample. Repeat the `file` part for several samples. Each sample must be at most 30 seconds and contain at least 4 seconds of effective single-speaker speech. See https://edgespeak.com/docs/api-audio#embeddings.",
        "requestBody": {
          "required": true,
          "content": {
            "multipart/form-data": {
              "schema": {
                "type": "object",
                "required": [
                  "file"
                ],
                "properties": {
                  "file": {
                    "type": "array",
                    "items": {
                      "type": "string",
                      "format": "binary"
                    },
                    "description": "Repeat this part once per sample."
                  },
                  "model": {
                    "type": "string",
                    "default": "EdgeSpeak/Lattice-1",
                    "description": "Speaker embedding model. Defaults to `EdgeSpeak/Lattice-1`. Embeddings are only comparable when they come from the same model."
                  }
                }
              },
              "encoding": {
                "file": {
                  "contentType": "application/octet-stream"
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "One entry per uploaded sample, in upload order. Full 256-value vectors are elided in this example.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/SpeakerEmbeddingList"
                },
                "example": {
                  "object": "list",
                  "model": "EdgeSpeak/Lattice-1",
                  "data": [
                    {
                      "index": 0,
                      "embedding": [
                        0.0123,
                        -0.0456
                      ],
                      "effective_speech_seconds": 5.2
                    },
                    {
                      "index": 1,
                      "embedding": [
                        0.0311,
                        -0.0092
                      ],
                      "effective_speech_seconds": 6.1
                    }
                  ]
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/speaker/similarity": {
      "post": {
        "operationId": "compareSpeakerEmbeddings",
        "tags": [
          "Transcription and speakers"
        ],
        "summary": "Compare the two embeddings",
        "description": "Returns the cosine similarity of two speaker embeddings, from -1 to 1. Your application decides how to use the score; the endpoint does not decide real-world identity. See https://edgespeak.com/docs/api-audio#similarity.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/SpeakerSimilarityRequest"
              },
              "example": {
                "model": "EdgeSpeak/Lattice-1",
                "embedding_a": [
                  0.0123,
                  -0.0456
                ],
                "embedding_b": [
                  0.0311,
                  -0.0092
                ]
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The similarity score.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/SpeakerSimilarityResult"
                },
                "example": {
                  "object": "speaker_similarity",
                  "model": "EdgeSpeak/Lattice-1",
                  "similarity": 0.82
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/audio/voices": {
      "get": {
        "operationId": "listVoices",
        "tags": [
          "Speech and voice library"
        ],
        "summary": "List voices and check model compatibility",
        "description": "Lists voices before synthesis. Pick a voice with `available: true` and inspect its per-model `compatibility`; availability alone does not guarantee compatibility with every speech model. See https://edgespeak.com/docs/api-speech#voices.",
        "responses": {
          "200": {
            "description": "The voice library. The example shows one user voice and illustrative IDs.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/VoiceList"
                },
                "example": {
                  "voices": [
                    {
                      "id": "user:00000000-0000-4000-8000-000000000001",
                      "names": {
                        "en-US": "My voice"
                      },
                      "descriptions": {},
                      "supported_languages": [
                        "en-US"
                      ],
                      "origin": "cloned",
                      "compatibility": [],
                      "available": true,
                      "created_by_user": true
                    }
                  ]
                }
              }
            }
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      },
      "post": {
        "operationId": "createVoice",
        "tags": [
          "Speech and voice library"
        ],
        "summary": "Create a reusable user voice",
        "description": "Saves a reusable voice in the service host's library from a short reference clip and its transcript. Do not supply `model`. Use the returned `voice.id` for synthesis or deletion. See https://edgespeak.com/docs/api-speech#add-voice.",
        "requestBody": {
          "required": true,
          "content": {
            "multipart/form-data": {
              "schema": {
                "type": "object",
                "required": [
                  "audio_sample",
                  "ref_text",
                  "name",
                  "consent"
                ],
                "properties": {
                  "audio_sample": {
                    "type": "string",
                    "format": "binary",
                    "description": "Required reference audio, up to 10 MiB. Use a short, clear clip."
                  },
                  "ref_text": {
                    "type": "string",
                    "description": "Transcript of the reference audio."
                  },
                  "name": {
                    "type": "string",
                    "description": "Display name for the voice, returned under `names`."
                  },
                  "consent": {
                    "type": "boolean",
                    "enum": [
                      true
                    ],
                    "description": "Required: `true`."
                  },
                  "language": {
                    "type": "string",
                    "default": "zh-CN",
                    "description": "Language of the reference audio, such as `zh-CN` or `en-US`. Omitting it stores the voice as `zh-CN`. See https://edgespeak.com/docs/languages#by-api for how each API uses language codes."
                  },
                  "speaker_description": {
                    "type": "string",
                    "description": "Optional description of the speaker, returned under `descriptions`."
                  }
                }
              },
              "encoding": {
                "audio_sample": {
                  "contentType": "application/octet-stream"
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The created voice. Inspect its compatibility before selecting a synthesis model.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/VoiceCreated"
                },
                "example": {
                  "success": true,
                  "voice": {
                    "id": "user:00000000-0000-4000-8000-000000000001",
                    "names": {
                      "en-US": "My voice"
                    },
                    "descriptions": {},
                    "supported_languages": [
                      "en-US"
                    ],
                    "origin": "cloned",
                    "compatibility": [],
                    "available": true,
                    "created_by_user": true
                  }
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/audio/speech": {
      "post": {
        "operationId": "createSpeech",
        "tags": [
          "Speech and voice library"
        ],
        "summary": "Generate speech as WAV",
        "description": "Synthesizes speech from text. `model`, `voice` and either `input` or a non-empty `segments[]` are required; obtain voice IDs from `GET /v1/audio/voices`. The default `stream_format: \"audio\"` streams binary `audio/wav`, and `sse` returns `speech.audio.delta` events followed by `speech.audio.done`, with base64 audio in each delta and the WAV header in the first one. Omit `sample_rate` for native-rate output: the desktop gateway accepts an explicit native rate only and returns 400 for streaming resampling, and headless hosts reject explicit values. Use only the controls the selected model supports. See https://edgespeak.com/docs/api-speech#speech and https://edgespeak.com/docs/api#speech-options.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": [
                  "model",
                  "voice"
                ],
                "properties": {
                  "model": {
                    "type": "string",
                    "description": "Speech model ID from the model catalog. One model serves the whole request, including every segment."
                  },
                  "voice": {
                    "type": "string",
                    "description": "Voice ID from `GET /v1/audio/voices`, such as `builtin:bright-girl` or a saved `user:<uuid>`. `builtin:auto` lets the model design a voice, which `Qwen/Qwen3-TTS-VoiceDesign` requires; CustomVoice models fall back to their first official voice, `Vivian`."
                  },
                  "input": {
                    "type": "string",
                    "description": "Supply `input` or a non-empty `segments[]`."
                  },
                  "segments": {
                    "type": "array",
                    "minItems": 1,
                    "items": {
                      "$ref": "#/components/schemas/SpeechSegment"
                    },
                    "description": "Multi-segment input, at least one entry. Each segment inherits the top-level voice and generation options unless it overrides them."
                  },
                  "response_format": {
                    "type": "string",
                    "enum": [
                      "wav"
                    ],
                    "description": "WAV is the supported response format."
                  },
                  "stream_format": {
                    "type": "string",
                    "enum": [
                      "audio",
                      "sse"
                    ],
                    "default": "audio",
                    "description": "`audio` streams the WAV bytes back directly; `sse` returns an event stream whose deltas carry base64 audio. Defaults to `audio`."
                  },
                  "language": {
                    "type": "string",
                    "description": "Language of the input text, such as `zh` or `en-US`. Omit it or send `auto` to choose automatically. Models without a language input still check it against their `supported_languages` and use it to pick reference audio. A language the model does not support returns 422 `language_unsupported`. See https://edgespeak.com/docs/languages#speech for supported languages."
                  },
                  "seed": {
                    "type": "integer",
                    "description": "Sampling seed. Send the same seed to reproduce a result; omit it for a fresh result each time."
                  },
                  "sample_rate": {
                    "type": "integer",
                    "description": "Omit for native-rate output."
                  },
                  "guidance_scale": {
                    "type": "number",
                    "description": "Classifier-free guidance strength. Higher values follow the voice and text more closely at the cost of naturalness. Models without a diffusion stage ignore it."
                  },
                  "inference_steps": {
                    "type": "integer",
                    "description": "Diffusion steps. More steps trade speed for quality. Models without a diffusion stage ignore it."
                  },
                  "temperature": {
                    "type": "number",
                    "description": "Sampling temperature, range 0.1-2. A value outside the range returns 400; models that do not sample ignore it."
                  },
                  "top_p": {
                    "type": "number",
                    "description": "Nucleus sampling cutoff, range 0.1-1. A value outside the range returns 400; models that do not sample ignore it."
                  },
                  "top_k": {
                    "type": "integer",
                    "description": "Number of sampling candidates, range 1-100. A value outside the range returns 400; models that do not sample ignore it."
                  },
                  "repetition_penalty": {
                    "type": "number",
                    "description": "Penalty on repeated tokens, starting at `1` for no penalty. The upper bound depends on the model and a value outside the range returns 400; models that do not sample ignore it."
                  },
                  "retry_badcase": {
                    "type": "boolean",
                    "description": "Let the model retry a generation it judges failed. `openbmb/VoxCPM2` only; other models ignore it."
                  },
                  "clone_recipe": {
                    "type": "string",
                    "description": "`openbmb/VoxCPM2` clone recipe, `controllable` or `ultimate`. Omit it to use the voice's saved preference. `ultimate` has no style control, so pair a non-empty `instructions` with `controllable`. Another model returns 400 `clone_recipe_requires_voxcpm2`, and `builtin:auto` returns 400 `clone_recipe_requires_saved_voice`."
                  },
                  "instructions": {
                    "type": [
                      "string",
                      "null"
                    ],
                    "description": "Natural-language speaking style under the OpenAI field name, such as a tone or an emotion. Omitted, `\"\"`, or `null` keeps the style saved with the voice; non-empty text uses that style for this request. With `voice: \"builtin:design\"` it is the required voice description. A model that does not support instructions returns 400 for non-empty text instead of dropping it. On `FireRedTeam/FireRedTTS3-Instruct` and `k2-fsa/OmniVoice` a description works only with `voice: \"builtin:design\"`; a specific voice plus non-empty text returns 400 `bad_request` (`errors.broadcast.instructUnsupported`). The removed `style_instruction` field returns 400 `bad_request`."
                  },
                  "disable_style": {
                    "type": "boolean",
                    "description": "`true` speaks this request without any style; `false` or omitted has no effect. Sending `true` with a non-empty `instructions` returns 400 `conflicting_speech_instructions`."
                  },
                  "speed": {
                    "type": "number",
                    "description": "Playback rate, where `1` is the model's natural speed. Models without a speed control ignore it."
                  },
                  "user_marks": {
                    "description": "Pronunciation and control input; shape depends on the model."
                  },
                  "phonetic_spans": {
                    "description": "Pronunciation and control input; shape depends on the model."
                  },
                  "control_policy": {
                    "description": "Pronunciation and control input; shape depends on the model."
                  }
                }
              },
              "example": {
                "model": "Qwen/Qwen3-TTS-0.6B-Base",
                "voice": "builtin:bright-girl",
                "input": "Hello world.",
                "response_format": "wav"
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "Generated audio. The default body is binary `audio/wav`; with `stream_format: \"sse\"` it is an event stream whose terminal event is `speech.audio.done`. Also handle an `error` event or a disconnection before done.",
            "content": {
              "audio/wav": {
                "schema": {
                  "type": "string",
                  "format": "binary"
                }
              },
              "text/event-stream": {
                "schema": {
                  "type": "string"
                },
                "example": "event: speech.audio.delta\ndata: {\"type\":\"speech.audio.delta\",\"audio\":\"<base64 WAV header and PCM bytes>\"}\n\nevent: speech.audio.done\ndata: {\"type\":\"speech.audio.done\",\"diagnostics\":{},\"usage\":{\"input_tokens\":0,\"output_tokens\":0,\"total_tokens\":0}}\n"
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          },
          "422": {
            "$ref": "#/components/responses/422"
          }
        }
      }
    },
    "/audio/speech/examples": {
      "get": {
        "operationId": "listSpeechExamples",
        "tags": [
          "Speech and voice library"
        ],
        "summary": "Find model-specific speech examples",
        "description": "Returns model-specific speech examples with applicability. Discover valid IDs and modes in the unfiltered catalog first; an invalid filter returns 400. Each example carries `applicable` and `reason`; only apply examples whose controls your selected model supports. See https://edgespeak.com/docs/api-speech#examples.",
        "parameters": [
          {
            "name": "model",
            "in": "query",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Filter by model."
          },
          {
            "name": "mode",
            "in": "query",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Filter by mode, for example `voice-design`."
          },
          {
            "name": "language",
            "in": "query",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Filter by language, for example `en-US`."
          },
          {
            "name": "id",
            "in": "query",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Filter by example ID."
          },
          {
            "name": "recipe",
            "in": "query",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Filter by recipe."
          }
        ],
        "responses": {
          "200": {
            "description": "The example catalog.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/SpeechExampleCatalog"
                },
                "example": {
                  "schema_version": "1",
                  "catalog_version": "2026-09-01",
                  "verified_at": "2026-09-01T00:00:00Z",
                  "resolved_language": "en-US",
                  "models": [
                    {
                      "id": "openbmb/VoxCPM2"
                    }
                  ],
                  "examples": [
                    {
                      "model": "openbmb/VoxCPM2",
                      "mode": "voice-design",
                      "language": "en-US",
                      "applicable": true,
                      "reason": null
                    }
                  ]
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/audio/voices/{voice_id}": {
      "delete": {
        "operationId": "deleteVoice",
        "tags": [
          "Speech and voice library"
        ],
        "summary": "Delete a user voice",
        "description": "Removes a user voice from the managed library. Only user voices can be deleted; built-in voices are immutable. See https://edgespeak.com/docs/api-speech#delete-voice.",
        "parameters": [
          {
            "name": "voice_id",
            "in": "path",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "The voice ID returned when the voice was created, such as `user:<uuid>`."
          }
        ],
        "responses": {
          "200": {
            "description": "The voice was deleted.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/VoiceDeleted"
                },
                "example": {
                  "success": true,
                  "deleted": {
                    "id": "user:00000000-0000-4000-8000-000000000001",
                    "name": "My voice"
                  }
                }
              }
            }
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          }
        }
      }
    },
    "/responses": {
      "post": {
        "operationId": "createResponse",
        "tags": [
          "Language and vision"
        ],
        "summary": "Generate with Responses",
        "description": "OpenAI Responses-compatible generation. Send `model` and `input`, which can be a string or supported structured input items. Read generated text from message content inside `output[]` rather than a universal top-level text field, and inspect item types instead of assuming `output[0]` is text. Adding `stream: true` returns SSE, typically `response.output_text.delta` and `response.completed`. Each endpoint requires a model whose `supported_endpoints` includes that exact path. These APIs forward the selected model's protocol, so optional parameters and response details can vary by local worker or remote provider. Local models keep inputs on the device; a configured remote model receives the request sent to it. See https://edgespeak.com/docs/api-language#responses.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": [
                  "model",
                  "input"
                ],
                "properties": {
                  "model": {
                    "type": "string",
                    "description": "Text model ID from the model catalog. Its `supported_endpoints` must contain this path."
                  },
                  "input": {
                    "description": "A string, or supported structured input items."
                  },
                  "max_output_tokens": {
                    "type": "integer",
                    "description": "Upper bound on generated tokens. Omit it to use the model's own limit."
                  },
                  "stream": {
                    "type": "boolean",
                    "description": "`true` returns Server-Sent Events instead of a single JSON body."
                  }
                }
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B",
                "input": "Say hello in one short sentence.",
                "max_output_tokens": 128
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The generated response, or an SSE stream when `stream: true`. Generated text, token counts and IDs vary.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/ResponsesResult"
                },
                "example": {
                  "id": "resp_example",
                  "object": "response",
                  "status": "completed",
                  "model": "Qwen/Qwen3.5-4B",
                  "output": [
                    {
                      "type": "message",
                      "role": "assistant",
                      "status": "completed",
                      "content": [
                        {
                          "type": "output_text",
                          "text": "Hello!"
                        }
                      ]
                    }
                  ]
                }
              },
              "text/event-stream": {
                "schema": {
                  "type": "string"
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/chat/completions": {
      "post": {
        "operationId": "createChatCompletion",
        "tags": [
          "Language and vision"
        ],
        "summary": "Generate with Chat Completions",
        "description": "OpenAI Chat Completions-compatible generation over `messages[]`. Read `choices[].message.content` for text; tool-capable responses can instead carry `tool_calls`. A content item can be `text`, `image_url` with a data URL, or `input_video` with base64 video bytes on macOS and Linux builds with native video support, which also requires `ffmpeg` and `ffprobe` on the service machine. Streaming uses `chat.completion.chunk` events terminated by `data: [DONE]`; do not apply that parser to the other protocols. Each endpoint requires a model whose `supported_endpoints` includes that exact path. These APIs forward the selected model's protocol, so optional parameters and response details can vary by local worker or remote provider. Local models keep inputs on the device; a configured remote model receives the request sent to it. See https://edgespeak.com/docs/api-language#chat, https://edgespeak.com/docs/api#video-input and https://edgespeak.com/docs/api#context-window.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": [
                  "model",
                  "messages"
                ],
                "properties": {
                  "model": {
                    "type": "string",
                    "description": "Text model ID from the model catalog. Its `supported_endpoints` must contain this path."
                  },
                  "messages": {
                    "type": "array",
                    "items": {
                      "$ref": "#/components/schemas/ChatMessage"
                    },
                    "description": "Conversation turns in order, oldest first."
                  },
                  "max_tokens": {
                    "type": "integer",
                    "description": "Upper bound on generated tokens. Omit it to use the model's own limit."
                  },
                  "stream": {
                    "type": "boolean",
                    "description": "`true` returns Server-Sent Events instead of a single JSON body."
                  }
                }
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B",
                "messages": [
                  {
                    "role": "user",
                    "content": "Say hello in one short sentence."
                  }
                ],
                "max_tokens": 128
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The completion, or an SSE stream when `stream: true`.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/ChatCompletion"
                },
                "example": {
                  "id": "chatcmpl_example",
                  "object": "chat.completion",
                  "model": "Qwen/Qwen3.5-4B",
                  "choices": [
                    {
                      "index": 0,
                      "message": {
                        "role": "assistant",
                        "content": "Hello!"
                      },
                      "finish_reason": "stop"
                    }
                  ]
                }
              },
              "text/event-stream": {
                "schema": {
                  "type": "string"
                },
                "example": "data: {\"id\":\"chatcmpl_example\",\"object\":\"chat.completion.chunk\",\"choices\":[{\"index\":0,\"delta\":{\"content\":\"Hello!\"},\"finish_reason\":null}]}\n\ndata: [DONE]\n"
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/messages": {
      "post": {
        "operationId": "createMessage",
        "tags": [
          "Language and vision"
        ],
        "summary": "Generate with Messages",
        "description": "Anthropic Messages-compatible generation. Supply `model`, `messages` and `max_tokens`. This endpoint accepts `x-api-key` as an alternative to Bearer authentication and forwards `anthropic-version` and `anthropic-beta` when present. Read text blocks inside `content[]`; this shape differs from both Responses and Chat Completions, and errors use the Anthropic error envelope. Streaming uses `content_block_delta` and `message_stop`. Each endpoint requires a model whose `supported_endpoints` includes that exact path. These APIs forward the selected model's protocol, so optional parameters and response details can vary by local worker or remote provider. Local models keep inputs on the device; a configured remote model receives the request sent to it. See https://edgespeak.com/docs/api-language#messages.",
        "parameters": [
          {
            "name": "anthropic-version",
            "in": "header",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Forwarded to the model when present, for example `2023-06-01`."
          },
          {
            "name": "anthropic-beta",
            "in": "header",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "Forwarded to the model when present."
          }
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": [
                  "model",
                  "messages",
                  "max_tokens"
                ],
                "properties": {
                  "model": {
                    "type": "string",
                    "description": "Text model ID from the model catalog. Its `supported_endpoints` must contain this path."
                  },
                  "messages": {
                    "type": "array",
                    "items": {
                      "$ref": "#/components/schemas/ChatMessage"
                    },
                    "description": "Conversation turns in order, oldest first."
                  },
                  "max_tokens": {
                    "type": "integer",
                    "description": "Required upper bound on generated tokens."
                  },
                  "stream": {
                    "type": "boolean",
                    "description": "`true` returns Server-Sent Events instead of a single JSON body."
                  }
                }
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B",
                "max_tokens": 128,
                "messages": [
                  {
                    "role": "user",
                    "content": "Say hello in one short sentence."
                  }
                ]
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The message, or an SSE stream when `stream: true`.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/AnthropicMessage"
                },
                "example": {
                  "id": "msg_example",
                  "type": "message",
                  "role": "assistant",
                  "model": "Qwen/Qwen3.5-4B",
                  "content": [
                    {
                      "type": "text",
                      "text": "Hello!"
                    }
                  ],
                  "stop_reason": "end_turn",
                  "usage": {
                    "input_tokens": 16,
                    "output_tokens": 3
                  }
                }
              },
              "text/event-stream": {
                "schema": {
                  "type": "string"
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/tokenize": {
      "post": {
        "operationId": "tokenizeText",
        "tags": [
          "Language and vision"
        ],
        "summary": "Tokenize text",
        "description": "Tokenizes text with the selected model's tokenizer. The local worker takes `content`; a remote provider may expose a different tokenization contract, which the gateway forwards. Token IDs are model-specific, so do not hard-code them. See https://edgespeak.com/docs/api-language#tokenize.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": [
                  "model",
                  "content"
                ],
                "properties": {
                  "model": {
                    "type": "string",
                    "description": "Model whose tokenizer is used. The same text yields different token IDs on different models."
                  },
                  "content": {
                    "type": "string",
                    "description": "Text to tokenize."
                  }
                }
              },
              "example": {
                "model": "Qwen/Qwen3.5-4B",
                "content": "Hello world."
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The token IDs. The array is illustrative; count its length when only an array is returned.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/TokenizeResult"
                },
                "example": {
                  "tokens": [
                    9707,
                    1879,
                    13
                  ]
                }
              }
            }
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          },
          "503": {
            "$ref": "#/components/responses/503"
          },
          "413": {
            "$ref": "#/components/responses/413"
          }
        }
      }
    },
    "/mcp": {
      "post": {
        "operationId": "callMcpTool",
        "tags": [
          "MCP tools"
        ],
        "summary": "Initialize and call a tool",
        "description": "Streamable HTTP JSON-RPC transport at the server root, not `/v1/mcp`. The service is stateless with JSON responses, so there is no session ID to save. Initialize first, then send the `notifications/initialized` notification and call tools with `tools/call`; reuse the negotiated version in the `MCP-Protocol-Version` header. Discover tool schemas with `tools/list`. Paths accepted by MCP are on the service host and returned artifact paths are not download URLs, so prefer inline audio or the multipart HTTP APIs from a remote client. See https://edgespeak.com/docs/mcp#http-example.",
        "servers": [
          {
            "url": "http://127.0.0.1:1117",
            "description": "Desktop app gateway"
          },
          {
            "url": "http://127.0.0.1:1118",
            "description": "Headless service started with edgespeak-cli serve"
          }
        ],
        "parameters": [
          {
            "name": "Accept",
            "in": "header",
            "required": true,
            "schema": {
              "type": "string",
              "default": "application/json, text/event-stream"
            },
            "description": "Streamable HTTP requires both media types."
          },
          {
            "name": "MCP-Protocol-Version",
            "in": "header",
            "required": false,
            "schema": {
              "type": "string"
            },
            "description": "The version negotiated by `initialize`. Send it on every request after initialization."
          }
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/JsonRpcRequest"
              },
              "example": {
                "jsonrpc": "2.0",
                "id": 2,
                "method": "tools/call",
                "params": {
                  "name": "edgespeak_segment_sentences",
                  "arguments": {
                    "text": "Hello world. Let us begin."
                  }
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "A JSON-RPC result. A tool failure can still arrive here, so inspect `error` and `result.isError` first.",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/JsonRpcResponse"
                },
                "example": {
                  "jsonrpc": "2.0",
                  "id": 2,
                  "result": {
                    "structuredContent": {
                      "metadata": {
                        "sentence_count": 2
                      },
                      "result": [
                        {
                          "text": "Hello world.",
                          "start": 0.0,
                          "end": 0.0
                        },
                        {
                          "text": "Let us begin.",
                          "start": 0.0,
                          "end": 0.0
                        }
                      ],
                      "truncated": false
                    }
                  }
                }
              }
            }
          },
          "202": {
            "description": "A notification such as `notifications/initialized` was accepted. The body is empty."
          },
          "400": {
            "$ref": "#/components/responses/400"
          },
          "401": {
            "$ref": "#/components/responses/401"
          },
          "403": {
            "$ref": "#/components/responses/403"
          }
        }
      }
    }
  },
  "components": {
    "securitySchemes": {
      "bearerAuth": {
        "type": "http",
        "scheme": "bearer",
        "description": "Send the gateway API Key as `Authorization: Bearer <key>`. The desktop key has the form `sk-edgespeak-` followed by 64 hexadecimal characters; a headless key is any non-empty string you set in `EDGESPEAK_API_KEY`."
      },
      "apiKeyAuth": {
        "type": "apiKey",
        "in": "header",
        "name": "x-api-key",
        "description": "Alternative header carrying the same gateway API Key."
      }
    },
    "responses": {
      "400": {
        "description": "Invalid request. Fix the request before retrying; use `error.code` for program logic and `error.param` to locate the input.",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      },
      "401": {
        "description": "Invalid or missing API key.",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      },
      "403": {
        "description": "Host or origin is not allowed, or the license was rejected. Inspect `error.code` to tell them apart.",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      },
      "404": {
        "description": "Unknown model ID (`model_not_found`).",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      },
      "413": {
        "description": "Request body exceeds 512 MiB.",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      },
      "422": {
        "description": "The selected model does not cover this language (`language_unsupported`). Change the model, not the language code.",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      },
      "503": {
        "description": "A required local model is still being prepared (`model_downloading`) or the service is busy (`service_busy`). Honor `Retry-After` when present.",
        "content": {
          "application/json": {
            "schema": {
              "$ref": "#/components/schemas/Error"
            }
          }
        }
      }
    },
    "schemas": {
      "Error": {
        "type": "object",
        "description": "Structured error envelope used by the gateway. `POST /v1/messages` and remote models may instead return the provider's own compatible envelope.",
        "required": [
          "error"
        ],
        "properties": {
          "error": {
            "type": "object",
            "required": [
              "message"
            ],
            "properties": {
              "message": {
                "type": "string",
                "description": "Human readable diagnostic. Not part of the stable contract; branch on `code` instead."
              },
              "type": {
                "type": "string",
                "description": "Error family, for example `invalid_request_error`."
              },
              "code": {
                "type": "string",
                "description": "Stable machine readable code, for example `bad_request`, `asr_language_unsupported`, `language_unsupported`, `model_capability_unsupported`, `model_not_found`, `model_downloading`, `service_busy`, `alignment_failed`, `alignment_search_budget_exceeded`."
              },
              "param": {
                "type": [
                  "string",
                  "null"
                ],
                "description": "Request field that caused the failure, when one can be identified."
              },
              "alignment": {
                "type": "object",
                "description": "Only on `alignment_failed` and `alignment_search_budget_exceeded`: the facts of the failed alignment. See https://edgespeak.com/docs/api-align#alignment-failure.",
                "properties": {
                  "reason": {
                    "type": "string",
                    "enum": [
                      "no_path",
                      "no_words",
                      "collapsed",
                      "search_incomplete"
                    ]
                  },
                  "search_effort_used": {
                    "type": "string",
                    "enum": [
                      "standard",
                      "extended"
                    ]
                  },
                  "search_path": {
                    "type": "string",
                    "enum": [
                      "whole_audio",
                      "streaming"
                    ]
                  },
                  "search_limit": {
                    "type": "string",
                    "enum": [
                      "none",
                      "budget",
                      "graph_size",
                      "estimate_overflow",
                      "streaming"
                    ]
                  },
                  "retry_recommended": {
                    "type": "boolean"
                  },
                  "estimated_extra_bytes": {
                    "type": "integer"
                  }
                },
                "required": [
                  "retry_recommended"
                ]
              }
            }
          }
        },
        "example": {
          "error": {
            "message": "'threshold' must be within [0, 1]",
            "type": "invalid_request_error",
            "code": "bad_request",
            "param": "threshold"
          }
        }
      },
      "Usage": {
        "type": "object",
        "description": "Audio duration accounting for transcription and alignment.",
        "properties": {
          "type": {
            "type": "string",
            "enum": [
              "duration"
            ]
          },
          "seconds": {
            "type": "number",
            "description": "Seconds of audio processed. For a windowed alignment this is the processed window, not the full media duration."
          }
        }
      },
      "Word": {
        "type": "object",
        "description": "One word with its timing. All timing values are seconds.",
        "properties": {
          "word": {
            "type": "string"
          },
          "start": {
            "type": "number"
          },
          "end": {
            "type": "number"
          },
          "score": {
            "type": "number",
            "minimum": 0,
            "maximum": 1,
            "description": "Confidence in [0, 1]."
          }
        }
      },
      "Segment": {
        "type": "object",
        "description": "Transcript or alignment segment. Optional keys are omitted when unavailable and key order is not guaranteed.",
        "properties": {
          "id": {
            "type": "integer",
            "description": "Sequential index of the segment within the result."
          },
          "start": {
            "type": "number",
            "description": "Segment start in seconds."
          },
          "end": {
            "type": "number",
            "description": "Segment end in seconds."
          },
          "text": {
            "type": "string",
            "description": "Segment text."
          },
          "speaker": {
            "type": [
              "string",
              "null"
            ],
            "description": "Speaker label, or `null` when no speaker was assigned."
          },
          "words": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/Word"
            },
            "description": "Word timings. Present only when word timestamps were requested."
          }
        }
      },
      "DiarizedSegment": {
        "type": "object",
        "description": "Segment shape returned by `response_format=diarized_json`. The identifier is a string such as `seg_0` and the speaker label can be null.",
        "properties": {
          "type": {
            "type": "string",
            "enum": [
              "transcript.text.segment"
            ]
          },
          "id": {
            "type": "string"
          },
          "start": {
            "type": "number"
          },
          "end": {
            "type": "number"
          },
          "text": {
            "type": "string"
          },
          "speaker": {
            "type": [
              "string",
              "null"
            ]
          },
          "words": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/Word"
            }
          }
        }
      },
      "TranscriptJson": {
        "type": "object",
        "description": "Shape returned by `response_format=json`.",
        "properties": {
          "text": {
            "type": "string"
          },
          "start": {
            "type": "number"
          },
          "duration": {
            "type": "number"
          },
          "language": {
            "type": "string"
          },
          "usage": {
            "$ref": "#/components/schemas/Usage"
          }
        }
      },
      "TranscriptVerbose": {
        "type": "object",
        "description": "Shape returned by `response_format=verbose_json`.",
        "properties": {
          "task": {
            "type": "string",
            "enum": [
              "transcribe"
            ]
          },
          "duration": {
            "type": "number"
          },
          "language": {
            "type": "string"
          },
          "text": {
            "type": "string"
          },
          "segments": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/Segment"
            }
          },
          "word_timestamps_unavailable": {
            "type": "array",
            "description": "Segments whose alignment failed. Their text is kept but they carry no words, and their times are the original segment window. Each item is `{start, end}` in seconds. Omitted when every segment aligned.",
            "items": {
              "type": "object",
              "properties": {
                "start": {
                  "type": "number"
                },
                "end": {
                  "type": "number"
                }
              }
            }
          },
          "usage": {
            "$ref": "#/components/schemas/Usage"
          }
        }
      },
      "TranscriptDiarized": {
        "type": "object",
        "description": "Shape returned by `response_format=diarized_json`.",
        "properties": {
          "text": {
            "type": "string"
          },
          "segments": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/DiarizedSegment"
            }
          },
          "word_timestamps_unavailable": {
            "type": "array",
            "description": "Segments whose alignment failed. Their text is kept but they carry no words, and their times are the original segment window. Each item is `{start, end}` in seconds. Omitted when every segment aligned.",
            "items": {
              "type": "object",
              "properties": {
                "start": {
                  "type": "number"
                },
                "end": {
                  "type": "number"
                }
              }
            }
          },
          "usage": {
            "$ref": "#/components/schemas/Usage"
          }
        }
      },
      "AlignmentResult": {
        "type": "object",
        "description": "Forced alignment result. Timestamps stay relative to the original audio.",
        "properties": {
          "task": {
            "type": "string",
            "enum": [
              "align"
            ]
          },
          "duration": {
            "type": "number",
            "description": "Full media duration."
          },
          "language": {
            "type": "string",
            "description": "Canonical language code, when the runtime resolved one."
          },
          "language_candidates": {
            "type": "array",
            "items": {
              "type": "string"
            },
            "description": "Ordered candidate languages the alignment actually used, preferred first. Comes from the request `language` or from the reference text; it routes pronunciation and is not an acoustic check of the audio language. Omitted when the candidate list is empty, for example when the request passes `und` or no language can be resolved."
          },
          "text": {
            "type": "string"
          },
          "segments": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/Segment"
            }
          },
          "normalization_regions": {
            "type": "array",
            "items": {
              "type": "object"
            },
            "description": "Regions and candidates produced by text normalization. Word-level `provenance` carries `source_start_byte`, `source_end_byte` and `normalizations[]`, relative to the original UTF-8 transcript."
          },
          "search_effort_used": {
            "type": "string",
            "enum": [
              "standard",
              "extended"
            ],
            "description": "The search that actually ran. Present when the service reports it."
          },
          "search_retried": {
            "type": "boolean",
            "description": "`true` when `search_effort=auto` actually ran the wider search after the first attempt failed. Omitted otherwise."
          },
          "usage": {
            "$ref": "#/components/schemas/Usage"
          }
        }
      },
      "SegmentationResult": {
        "type": "object",
        "description": "Sentence or paragraph segmentation result. Plain text input yields no timestamps or speakers; timed input can preserve them.",
        "properties": {
          "task": {
            "type": "string",
            "enum": [
              "segment"
            ]
          },
          "text": {
            "type": "string",
            "description": "In paragraph mode this joins paragraphs with a blank line."
          },
          "segments": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/Segment"
            }
          }
        }
      },
      "NormalizationSpan": {
        "type": "object",
        "description": "One replaced region. Offsets are half-open UTF-8 byte ranges in the original input, not JavaScript character indices.",
        "properties": {
          "start_byte": {
            "type": "integer"
          },
          "end_byte": {
            "type": "integer"
          },
          "class": {
            "type": "string",
            "description": "Normalization category, for example `money` or `time`."
          },
          "source": {
            "type": "string"
          },
          "output": {
            "type": "string"
          }
        }
      },
      "NormalizationAlternative": {
        "type": "object",
        "properties": {
          "rank": {
            "type": "integer"
          },
          "text": {
            "type": "string",
            "description": "The converted paragraph."
          },
          "spans": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/NormalizationSpan"
            }
          }
        }
      },
      "NormalizationResult": {
        "type": "object",
        "properties": {
          "task": {
            "type": "string",
            "enum": [
              "normalize"
            ]
          },
          "model": {
            "type": "string"
          },
          "language": {
            "type": "string"
          },
          "mode": {
            "type": "string",
            "enum": [
              "tn",
              "itn"
            ]
          },
          "text": {
            "type": "string",
            "description": "Preserves the whole input."
          },
          "alternatives": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/NormalizationAlternative"
            }
          }
        }
      },
      "ModelCatalogEntry": {
        "type": "object",
        "description": "One catalog entry. Entries appear whether or not the model is downloaded or loaded.",
        "properties": {
          "id": {
            "type": "string",
            "description": "Canonical model ID. Use it verbatim in requests."
          },
          "object": {
            "type": "string",
            "enum": [
              "model"
            ]
          },
          "created": {
            "type": "integer"
          },
          "owned_by": {
            "type": "string"
          },
          "supported_endpoints": {
            "type": "array",
            "items": {
              "type": "string"
            },
            "description": "Paths this model can serve, for example `/v1/chat/completions` or `/v1/audio/alignments`."
          },
          "features": {
            "type": "array",
            "items": {
              "type": "string"
            },
            "description": "Capability flags such as `reasoning`, `vision` or `tool_calling`."
          },
          "execution_location": {
            "type": "string",
            "enum": [
              "local",
              "remote"
            ],
            "description": "`local` executes on the service host; a configured `remote` model receives the request sent to it."
          },
          "default_for": {
            "type": "array",
            "items": {
              "type": "string"
            },
            "description": "Paths for which this model is the default. Read it instead of hard-coding aliases."
          },
          "supported_languages": {
            "type": "array",
            "items": {
              "type": "string"
            },
            "description": "Speech models: languages you can pass as `language` to `/v1/audio/speech`, as canonical codes (FireRedTTS3 dialects use `zh-x-<name>`). Region and script variants of a listed language, such as `zh-TW`, are accepted. `auto` is always accepted."
          },
          "x_edgespeak": {
            "type": "object",
            "description": "EdgeSpeak extensions. On speech models, `speech_language.open_set` is `true` when the model also accepts ISO 639 codes outside `supported_languages`, such as `k2-fsa/OmniVoice`.",
            "properties": {
              "speech_language": {
                "type": "object",
                "properties": {
                  "open_set": {
                    "type": "boolean"
                  }
                }
              }
            }
          }
        }
      },
      "ModelList": {
        "type": "object",
        "properties": {
          "object": {
            "type": "string",
            "enum": [
              "list"
            ]
          },
          "data": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/ModelCatalogEntry"
            }
          }
        }
      },
      "ModelLifecycleRequest": {
        "type": "object",
        "description": "All four POST lifecycle operations accept exactly this body; extra fields are rejected.",
        "additionalProperties": false,
        "required": [
          "model"
        ],
        "properties": {
          "model": {
            "type": "string",
            "description": "Catalog model ID."
          }
        }
      },
      "DownloadJob": {
        "type": "object",
        "properties": {
          "model": {
            "type": "string"
          },
          "state": {
            "type": "string",
            "enum": [
              "downloading",
              "completed",
              "failed",
              "cancelled"
            ]
          },
          "bytes_downloaded": {
            "type": "integer"
          },
          "total_bytes": {
            "type": "integer",
            "description": "`0` can mean the total is not known yet."
          },
          "error": {
            "type": [
              "string",
              "null"
            ]
          }
        }
      },
      "DownloadJobList": {
        "type": "object",
        "properties": {
          "object": {
            "type": "string",
            "enum": [
              "list"
            ]
          },
          "data": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/DownloadJob"
            }
          }
        }
      },
      "DownloadCancelResult": {
        "type": "object",
        "properties": {
          "model": {
            "type": "string"
          },
          "state": {
            "type": "string",
            "description": "`cancelling` while the asynchronous cancellation runs."
          }
        }
      },
      "ModelLifecycleResult": {
        "type": "object",
        "description": "Stable core fields. Host-specific fields may also appear.",
        "properties": {
          "success": {
            "type": "boolean"
          },
          "model": {
            "type": "string"
          },
          "status": {
            "type": "string",
            "enum": [
              "loaded",
              "unloaded"
            ]
          }
        }
      },
      "QueueStatus": {
        "type": "object",
        "description": "One queue. `active` includes admitted work that is still preparing or cleaning up, so counts from the admission and model layers must not be summed.",
        "properties": {
          "active": {
            "type": "integer"
          },
          "waiting": {
            "type": "integer"
          },
          "execution_capacity": {
            "type": "integer"
          },
          "oldest_wait_ms": {
            "type": "integer"
          }
        }
      },
      "RunningModels": {
        "type": "object",
        "properties": {
          "object": {
            "type": "string",
            "enum": [
              "list"
            ]
          },
          "data": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "id": {
                  "type": "string"
                },
                "status": {
                  "type": "object",
                  "description": "Model-specific runtime status."
                }
              }
            }
          },
          "runtime": {
            "type": "object",
            "properties": {
              "state": {
                "type": "string",
                "description": "`running`, or `loading` during a cold load or a model switch, when the model list is temporarily empty."
              },
              "queues": {
                "type": "object",
                "description": "Optional; present only on builds containing the queue-status change. Admission queues are `http_transcribe`, `http_alignment`, `http_segmentation`, `http_normalization`, `http_diarization`, `http_speaker_embedding`, `http_broadcast` and `http_realtime`; `native_transcribe` and `native_broadcast` describe model ownership across REST, Realtime, MCP and preloading.",
                "additionalProperties": {
                  "$ref": "#/components/schemas/QueueStatus"
                }
              }
            }
          }
        }
      },
      "HealthStatus": {
        "type": "object",
        "description": "A running service can return HTTP 200 with an inactive license, and a healthy service is not the same as a loaded model.",
        "properties": {
          "status": {
            "type": "string",
            "enum": [
              "ok"
            ]
          },
          "license": {
            "type": "string",
            "enum": [
              "active",
              "inactive"
            ]
          }
        }
      },
      "Voice": {
        "type": "object",
        "description": "One voice record. Names and descriptions are locale maps, not a single name string.",
        "properties": {
          "id": {
            "type": "string",
            "description": "Voice ID, for example `builtin:bright-girl` or `user:<uuid>`."
          },
          "names": {
            "type": "object",
            "additionalProperties": {
              "type": "string"
            }
          },
          "descriptions": {
            "type": "object",
            "additionalProperties": {
              "type": "string"
            }
          },
          "supported_languages": {
            "type": "array",
            "items": {
              "type": "string"
            }
          },
          "origin": {
            "type": "string",
            "description": "For example `cloned`."
          },
          "compatibility": {
            "type": "array",
            "items": {
              "type": "object"
            },
            "description": "Per-model compatibility. Availability alone does not guarantee compatibility with every speech model."
          },
          "available": {
            "type": "boolean"
          },
          "created_by_user": {
            "type": "boolean",
            "description": "Only user voices can be deleted; built-in voices are immutable."
          }
        }
      },
      "VoiceList": {
        "type": "object",
        "properties": {
          "voices": {
            "type": "array",
            "items": {
              "$ref": "#/components/schemas/Voice"
            }
          }
        }
      },
      "VoiceCreated": {
        "type": "object",
        "properties": {
          "success": {
            "type": "boolean"
          },
          "voice": {
            "$ref": "#/components/schemas/Voice"
          }
        }
      },
      "VoiceDeleted": {
        "type": "object",
        "properties": {
          "success": {
            "type": "boolean"
          },
          "deleted": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "name": {
                "type": "string"
              }
            }
          }
        }
      },
      "SpeechExampleCatalog": {
        "type": "object",
        "description": "Model-specific speech examples. Only apply an example whose controls the selected model supports.",
        "properties": {
          "schema_version": {
            "type": "string"
          },
          "catalog_version": {
            "type": "string"
          },
          "verified_at": {
            "type": "string"
          },
          "resolved_language": {
            "type": "string"
          },
          "models": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "examples": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "model": {
                  "type": "string"
                },
                "mode": {
                  "type": "string",
                  "description": "For example `emotion-vector`, `style-control` or `voice-design`."
                },
                "language": {
                  "type": "string"
                },
                "id": {
                  "type": "string"
                },
                "recipe": {
                  "type": "string"
                },
                "applicable": {
                  "type": "boolean"
                },
                "reason": {
                  "type": [
                    "string",
                    "null"
                  ]
                }
              }
            }
          }
        }
      },
      "DiarizationResult": {
        "type": "object",
        "description": "Speaker activity timeline without text. Labels identify clusters within this recording, not personal identities.",
        "properties": {
          "duration": {
            "type": "number"
          },
          "speakers": {
            "type": "array",
            "items": {
              "type": "string"
            }
          },
          "segments": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "start": {
                  "type": "number"
                },
                "end": {
                  "type": "number"
                },
                "speaker": {
                  "type": "string"
                }
              }
            }
          }
        }
      },
      "SpeakerEmbeddingList": {
        "type": "object",
        "properties": {
          "object": {
            "type": "string",
            "enum": [
              "list"
            ]
          },
          "model": {
            "type": "string"
          },
          "data": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "index": {
                  "type": "integer"
                },
                "embedding": {
                  "type": "array",
                  "items": {
                    "type": "number"
                  },
                  "minItems": 256,
                  "maxItems": 256,
                  "description": "256-dimensional vector."
                },
                "effective_speech_seconds": {
                  "type": "number"
                }
              }
            }
          }
        }
      },
      "SpeakerSimilarityRequest": {
        "type": "object",
        "description": "Both vectors must have 256 finite values and a non-zero norm. The endpoint accepts neither `threshold` nor `match`.",
        "required": [
          "embedding_a",
          "embedding_b"
        ],
        "properties": {
          "embedding_a": {
            "type": "array",
            "items": {
              "type": "number"
            },
            "minItems": 256,
            "maxItems": 256,
            "description": "First embedding, exactly 256 numbers as returned by `POST /v1/speaker/embeddings`."
          },
          "embedding_b": {
            "type": "array",
            "items": {
              "type": "number"
            },
            "minItems": 256,
            "maxItems": 256,
            "description": "Second embedding, in the same shape and from the same model as `embedding_a`."
          },
          "model": {
            "type": "string",
            "default": "EdgeSpeak/Lattice-1",
            "description": "Model the two embeddings came from. Defaults to `EdgeSpeak/Lattice-1`."
          }
        }
      },
      "SpeakerSimilarityResult": {
        "type": "object",
        "properties": {
          "object": {
            "type": "string",
            "enum": [
              "speaker_similarity"
            ]
          },
          "model": {
            "type": "string"
          },
          "similarity": {
            "type": "number",
            "minimum": -1,
            "maximum": 1,
            "description": "Cosine similarity from -1 to 1. It does not decide real-world identity."
          }
        }
      },
      "TokenizeResult": {
        "type": "object",
        "description": "Token IDs depend on the selected model's tokenizer; do not hard-code them. Tokenizing raw text does not include the extra tokens a chat template may insert.",
        "properties": {
          "tokens": {
            "type": "array",
            "items": {
              "type": "integer"
            }
          }
        }
      },
      "ResponsesResult": {
        "type": "object",
        "description": "OpenAI Responses-compatible result. Read generated text from message content inside `output[]` rather than a top-level text field; reasoning-capable models may add reasoning items first.",
        "properties": {
          "id": {
            "type": "string"
          },
          "object": {
            "type": "string",
            "enum": [
              "response"
            ]
          },
          "status": {
            "type": "string"
          },
          "model": {
            "type": "string"
          },
          "output": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string"
                },
                "role": {
                  "type": "string"
                },
                "status": {
                  "type": "string"
                },
                "content": {
                  "type": "array",
                  "items": {
                    "type": "object",
                    "properties": {
                      "type": {
                        "type": "string"
                      },
                      "text": {
                        "type": "string"
                      }
                    }
                  }
                }
              }
            }
          }
        }
      },
      "ChatCompletion": {
        "type": "object",
        "description": "OpenAI Chat Completions-compatible result. Tool-capable responses can carry `tool_calls` instead of text.",
        "properties": {
          "id": {
            "type": "string"
          },
          "object": {
            "type": "string",
            "enum": [
              "chat.completion"
            ]
          },
          "model": {
            "type": "string"
          },
          "choices": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "index": {
                  "type": "integer"
                },
                "message": {
                  "type": "object",
                  "properties": {
                    "role": {
                      "type": "string"
                    },
                    "content": {
                      "type": [
                        "string",
                        "null"
                      ]
                    },
                    "tool_calls": {
                      "type": "array",
                      "items": {
                        "type": "object"
                      }
                    }
                  }
                },
                "finish_reason": {
                  "type": [
                    "string",
                    "null"
                  ]
                }
              }
            }
          }
        }
      },
      "AnthropicMessage": {
        "type": "object",
        "description": "Anthropic Messages-compatible result. This shape differs from both Responses and Chat Completions.",
        "properties": {
          "id": {
            "type": "string"
          },
          "type": {
            "type": "string",
            "enum": [
              "message"
            ]
          },
          "role": {
            "type": "string"
          },
          "model": {
            "type": "string"
          },
          "content": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "type": {
                  "type": "string"
                },
                "text": {
                  "type": "string"
                }
              }
            }
          },
          "stop_reason": {
            "type": [
              "string",
              "null"
            ]
          },
          "usage": {
            "type": "object",
            "properties": {
              "input_tokens": {
                "type": "integer"
              },
              "output_tokens": {
                "type": "integer"
              }
            }
          }
        }
      },
      "ChatMessage": {
        "type": "object",
        "required": [
          "role",
          "content"
        ],
        "properties": {
          "role": {
            "type": "string",
            "description": "Message role, such as `user` or `assistant`."
          },
          "content": {
            "oneOf": [
              {
                "type": "string"
              },
              {
                "type": "array",
                "items": {
                  "type": "object",
                  "description": "A content item: `text`, `image_url` with a data URL, or `input_video` carrying base64 video bytes.",
                  "properties": {
                    "type": {
                      "type": "string",
                      "enum": [
                        "text",
                        "image_url",
                        "input_video"
                      ]
                    },
                    "text": {
                      "type": "string"
                    },
                    "image_url": {
                      "type": "object",
                      "properties": {
                        "url": {
                          "type": "string"
                        }
                      }
                    },
                    "input_video": {
                      "type": "object",
                      "properties": {
                        "data": {
                          "type": "string",
                          "description": "Base64-encoded video file bytes."
                        }
                      }
                    }
                  }
                }
              }
            ],
            "description": "Message text, or an array of content items for multimodal input."
          }
        }
      },
      "SpeechSegment": {
        "type": "object",
        "description": "One segment of a multi-segment speech request. Segments share the top-level model.",
        "properties": {
          "input": {
            "type": "string",
            "description": "Text for this segment."
          },
          "voice": {
            "type": "string",
            "description": "Voice for this segment. Omit it to inherit the top-level `voice`."
          },
          "instructions": {
            "type": [
              "string",
              "null"
            ],
            "description": "Speaking style for this segment. Non-empty text overrides the top-level style; omitted, `\"\"`, or `null` inherits it."
          },
          "disable_style": {
            "type": "boolean",
            "description": "`true` speaks this segment without any style; `false` or omitted inherits the top-level style. Cannot be combined with a non-empty `instructions`."
          },
          "speed": {
            "type": "number",
            "description": "Playback rate for this segment. Omit it to inherit the top-level `speed`."
          },
          "language": {
            "type": "string",
            "description": "Language for this segment. Omit it to inherit the top-level `language`."
          }
        }
      },
      "JsonRpcRequest": {
        "type": "object",
        "description": "A JSON-RPC 2.0 request or notification. A notification omits `id` and is answered with 202 and an empty body.",
        "required": [
          "jsonrpc",
          "method"
        ],
        "properties": {
          "jsonrpc": {
            "type": "string",
            "enum": [
              "2.0"
            ],
            "description": "Protocol version; always `2.0`."
          },
          "id": {
            "oneOf": [
              {
                "type": "integer"
              },
              {
                "type": "string"
              }
            ],
            "description": "Identifier echoed back in the response. Omit it to send a notification."
          },
          "method": {
            "type": "string",
            "description": "For example `initialize`, `notifications/initialized`, `tools/list` or `tools/call`."
          },
          "params": {
            "type": "object",
            "description": "For `tools/call`, `{name, arguments}`."
          }
        }
      },
      "JsonRpcResponse": {
        "type": "object",
        "description": "A JSON-RPC 2.0 response. A tool failure can still arrive with HTTP 200, so inspect `error` and `result.isError` before reading structured content.",
        "properties": {
          "jsonrpc": {
            "type": "string",
            "enum": [
              "2.0"
            ]
          },
          "id": {
            "oneOf": [
              {
                "type": "integer"
              },
              {
                "type": "string"
              }
            ]
          },
          "result": {
            "type": "object"
          },
          "error": {
            "type": "object",
            "properties": {
              "code": {
                "type": "integer"
              },
              "message": {
                "type": "string"
              },
              "data": {}
            }
          }
        }
      }
    }
  }
}
