Module livekit.plugins.microsoft_ai
Microsoft AI STT and TTS using explicit, deployment-specific endpoints.
TTS uses Azure Speech; STT uses the transcription WebSocket API. See the package README for supported contracts and the limits of the bounded live smoke coverage.
Classes
class STT (*,
vad: vad.VAD | None,
url: str | None = None,
model: str | None = None,
api_key: str | None = None,
auth_header: "Literal['Authorization', 'api-key'] | None" = None,
headers: Mapping[str, str] | None = None,
language: str | None = None,
http_session: aiohttp.ClientSession | None = None,
env_file: str | Path | None = None,
max_buffered_audio: float = 5.0)-
Expand source code
class STT(stt.STT): """Native streaming Microsoft AI transcription with explicit client commits. See the package README for deployment-specific contract and validation limits. Args: vad: A VAD emitting ordered INFERENCE_DONE events even during silence, with input-relative timestamps, and START_OF_SPEECH frames containing the detected onset/prefix through that timestamp (e.g. LiveKit's Silero VAD). Idle audio is withheld; these public frames recover the onset even when the start notification is delayed. Required explicitly: pass None only when driving flush/end_input yourself. AgentSession's separate VAD does not commit a native STT stream. url: Full WebSocket URL, or MICROSOFT_AI_STT_URL. No paths/query parameters are added automatically. model: Deployment model ID, or MICROSOFT_AI_STT_MODEL. No model is assumed. api_key: Credential, or MICROSOFT_AI_STT_API_KEY, sent using auth_header. auth_header: Authorization (the default, with Bearer prefix) or api-key (raw credential), or MICROSOFT_AI_STT_AUTH_HEADER. No auth fallback occurs. headers: Explicit authentication headers instead of api_key/auth_header environment lookup. Cannot be combined with either constructor argument. language: Optional transcription language hint, or MICROSOFT_AI_STT_LANGUAGE. http_session: Optional caller-owned aiohttp session. env_file: Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor arguments override environment variables, which override this file. max_buffered_audio: Client-side queued-audio and VAD start-prefix limits in seconds. Queue overflow or an oversized VAD prefix fails explicitly instead of silently dropping speech. """ def __init__( self, *, vad: vad.VAD | None, url: str | None = None, model: str | None = None, api_key: str | None = None, auth_header: Literal["Authorization", "api-key"] | None = None, headers: Mapping[str, str] | None = None, language: str | None = None, http_session: aiohttp.ClientSession | None = None, env_file: str | Path | None = None, max_buffered_audio: float = 5.0, ) -> None: super().__init__( capabilities=stt.STTCapabilities( streaming=True, interim_results=True, offline_recognize=False ) ) config = Configuration(env_file) positive_timeout(max_buffered_audio, "max_buffered_audio") if language is None: language = config.get("MICROSOFT_AI_STT_LANGUAGE") or None if language is not None and not language.strip(): raise ValueError("language must be nonempty when supplied") self._model = config.required(model, "MICROSOFT_AI_STT_MODEL") self._client = HTTPClient( config=config, service="STT", url=url, api_key=api_key, headers=headers, http_session=http_session, auth_header=auth_header, ) self._vad = vad self._language = language self._max_buffered_audio = max_buffered_audio self._streams: weakref.WeakSet[SpeechStream] = weakref.WeakSet() self._closed = False @property def model(self) -> str: return self._model @property def provider(self) -> str: return "Microsoft AI" async def _recognize_impl( self, buffer: utils.AudioBuffer, *, language: NotGivenOr[str] = NOT_GIVEN, conn_options: APIConnectOptions, ) -> stt.SpeechEvent: raise NotImplementedError("Microsoft AI STT supports stream(), not batch recognize()") def stream( self, *, language: NotGivenOr[str] = NOT_GIVEN, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS, ) -> SpeechStream: """Open a transcription stream; flush commits a turn, end_input also drains finals.""" if self._closed: raise RuntimeError("Microsoft AI STT is closed") positive_timeout(conn_options.timeout, "conn_options.timeout") selected_language = language if utils.is_given(language) else self._language if selected_language is not None and not selected_language.strip(): raise ValueError("language must be nonempty when supplied") stream = SpeechStream(stt=self, language=selected_language, conn_options=conn_options) self._streams.add(stream) return stream async def aclose(self) -> None: self._closed = True await asyncio.gather(*(stream.aclose() for stream in list(self._streams))) await self._client.aclose()Native streaming Microsoft AI transcription with explicit client commits.
See the package README for deployment-specific contract and validation limits.
Args
vad- A VAD emitting ordered INFERENCE_DONE events even during silence, with input-relative timestamps, and START_OF_SPEECH frames containing the detected onset/prefix through that timestamp (e.g. LiveKit's Silero VAD). Idle audio is withheld; these public frames recover the onset even when the start notification is delayed. Required explicitly: pass None only when driving flush/end_input yourself. AgentSession's separate VAD does not commit a native STT stream.
url- Full WebSocket URL, or MICROSOFT_AI_STT_URL. No paths/query parameters are added automatically.
model- Deployment model ID, or MICROSOFT_AI_STT_MODEL. No model is assumed.
api_key- Credential, or MICROSOFT_AI_STT_API_KEY, sent using auth_header.
auth_header- Authorization (the default, with Bearer prefix) or api-key (raw credential), or MICROSOFT_AI_STT_AUTH_HEADER. No auth fallback occurs.
headers- Explicit authentication headers instead of api_key/auth_header environment lookup. Cannot be combined with either constructor argument.
language- Optional transcription language hint, or MICROSOFT_AI_STT_LANGUAGE.
http_session- Optional caller-owned aiohttp session.
env_file- Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor arguments override environment variables, which override this file.
max_buffered_audio- Client-side queued-audio and VAD start-prefix limits in seconds. Queue overflow or an oversized VAD prefix fails explicitly instead of silently dropping speech.
Ancestors
- livekit.agents.stt.stt.STT
- abc.ABC
- EventEmitter
- typing.Generic
Instance variables
prop model : str-
Expand source code
@property def model(self) -> str: return self._modelGet the model name/identifier for this STT instance.
Returns
The model name if available, "unknown" otherwise.
Note
Plugins should override this property to provide their model information.
prop provider : str-
Expand source code
@property def provider(self) -> str: return "Microsoft AI"Get the provider name/identifier for this STT instance.
Returns
The provider name if available, "unknown" otherwise.
Note
Plugins should override this property to provide their provider information.
Methods
async def aclose(self) ‑> None-
Expand source code
async def aclose(self) -> None: self._closed = True await asyncio.gather(*(stream.aclose() for stream in list(self._streams))) await self._client.aclose()Close the STT, and every stream/requests associated with it
def stream(self,
*,
language: NotGivenOr[str] = NOT_GIVEN,
conn_options: APIConnectOptions = APIConnectOptions(max_retry=3, retry_interval=2.0, timeout=10.0)) ‑> livekit.plugins.microsoft_ai.stt.SpeechStream-
Expand source code
def stream( self, *, language: NotGivenOr[str] = NOT_GIVEN, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS, ) -> SpeechStream: """Open a transcription stream; flush commits a turn, end_input also drains finals.""" if self._closed: raise RuntimeError("Microsoft AI STT is closed") positive_timeout(conn_options.timeout, "conn_options.timeout") selected_language = language if utils.is_given(language) else self._language if selected_language is not None and not selected_language.strip(): raise ValueError("language must be nonempty when supplied") stream = SpeechStream(stt=self, language=selected_language, conn_options=conn_options) self._streams.add(stream) return streamOpen a transcription stream; flush commits a turn, end_input also drains finals.
Inherited members
class TTS (*,
url: str | None = None,
region: str | None = None,
model: str | None = None,
voice: str | None = None,
language: str = 'en-US',
sample_rate: int | None = None,
api_key: str | None = None,
headers: Mapping[str, str] | None = None,
http_session: aiohttp.ClientSession | None = None,
env_file: str | Path | None = None,
request_timeout: float = 30.0,
max_text_length: int = 4096,
max_audio_bytes: int = 10485760)-
Expand source code
class TTS(tts.TTS): """Microsoft AI voices through Azure Speech's SSML REST API. Args: url: Full synthesis POST endpoint, or MICROSOFT_AI_TTS_URL. Its host/path are used exactly as supplied and take precedence over region. region: Public-cloud Azure Speech region, or MICROSOFT_AI_TTS_REGION. Used only when the URL is unset, not when it is empty or whitespace. model: Voice model, or MICROSOFT_AI_TTS_MODEL, e.g. MAI-Voice-2-Flash. Appended to a short voice name; optional when voice is a full ID, which it must then match (case-insensitively), not an alias. voice: Voice name such as en-US-Harper, or a full Azure Speech voice ID such as en-US-Harper:MAI-Voice-2-Flash, or MICROSOFT_AI_TTS_VOICE. language: SSML language, default en-US. sample_rate: Expected WAV sample rate, or MICROSOFT_AI_TTS_SAMPLE_RATE. api_key: Azure Speech subscription key, or MICROSOFT_AI_TTS_API_KEY. headers: Explicit headers instead of api_key/environment lookup. http_session: Optional caller-owned aiohttp session. env_file: Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor arguments override environment variables, which override this file. request_timeout: Total request/body deadline in seconds. max_text_length: Client-side character limit, not a claimed provider limit. max_audio_bytes: Maximum complete WAV response size in bytes. AgentSession wraps this nonstreaming provider in its existing sentence StreamAdapter. Native text/audio streaming and JSON/base64 responses are unsupported. """ def __init__( self, *, url: str | None = None, region: str | None = None, model: str | None = None, voice: str | None = None, language: str = "en-US", sample_rate: int | None = None, api_key: str | None = None, headers: Mapping[str, str] | None = None, http_session: aiohttp.ClientSession | None = None, env_file: str | Path | None = None, request_timeout: float = 30.0, max_text_length: int = 4096, max_audio_bytes: int = 10 * 1024 * 1024, ) -> None: config = Configuration(env_file) if sample_rate is None: try: sample_rate = int(config.get("MICROSOFT_AI_TTS_SAMPLE_RATE") or "") except ValueError: raise ValueError("Set MICROSOFT_AI_TTS_SAMPLE_RATE or pass sample_rate") from None if ( isinstance(sample_rate, bool) or not isinstance(sample_rate, int) or sample_rate not in _WAV_FORMATS ): raise ValueError("sample_rate must be one of 8000, 22050, 24000, 44100, 48000") for name, value in ( ("max_text_length", max_text_length), ("max_audio_bytes", max_audio_bytes), ): if isinstance(value, bool) or not isinstance(value, int) or value <= 0: raise ValueError(f"{name} must be a positive integer") positive_timeout(request_timeout, "request_timeout") super().__init__( capabilities=tts.TTSCapabilities(streaming=False), sample_rate=sample_rate, num_channels=1, ) voice = config.required(voice, "MICROSOFT_AI_TTS_VOICE") voice_name, separator, voice_model = voice.rpartition(":") if separator: if not voice_name or not voice_model: raise ValueError("voice must be a voice name or a full voice ID such as name:model") if model is None: model = config.get("MICROSOFT_AI_TTS_MODEL") if model is None: model = voice_model model = config.required(model, "MICROSOFT_AI_TTS_MODEL") if voice_model.casefold() != model.casefold(): raise ValueError("full voice ID must end in the configured model name") else: model = config.required(model, "MICROSOFT_AI_TTS_MODEL") voice = f"{voice}:{model}" if re.fullmatch(r"[A-Za-z]{2,3}(?:-[A-Za-z0-9]{2,8})*", language) is None: raise ValueError("language must be a language tag such as en-US") self._opts = _TTSOptions( model=model, voice=voice, sample_rate=sample_rate, language=language ) self._client = HTTPClient( config=config, service="TTS", url=_endpoint(config, url, region), api_key=api_key, headers=headers, http_session=http_session, ) self._client.headers = { key: value for key, value in self._client.headers.items() if key.lower() not in ("accept", "content-type", "x-microsoft-outputformat") } self._client.headers.update( { "Accept": "audio/wav", "Content-Type": "application/ssml+xml", "X-Microsoft-OutputFormat": _WAV_FORMATS[sample_rate], } ) self._request_timeout = request_timeout self._max_text_length = max_text_length self._max_audio_bytes = max_audio_bytes self._streams: weakref.WeakSet[ChunkedStream] = weakref.WeakSet() self._closed = False @property def model(self) -> str: return self._opts.model @property def provider(self) -> str: return "Microsoft AI" def synthesize( self, text: str, *, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS ) -> ChunkedStream: if self._closed: raise RuntimeError("Microsoft AI TTS is closed") if not text.strip(): raise ValueError("Microsoft AI TTS requires nonempty text") if len(text) > self._max_text_length: raise ValueError("Microsoft AI TTS text exceeds max_text_length") if re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f\ud800-\udfff\ufffe\uffff]", text): raise ValueError("Microsoft AI TTS text contains characters invalid in XML") positive_timeout(conn_options.timeout, "conn_options.timeout") stream = ChunkedStream(tts=self, input_text=text, conn_options=conn_options) self._streams.add(stream) return stream async def aclose(self) -> None: self._closed = True await asyncio.gather(*(stream.aclose() for stream in list(self._streams))) await self._client.aclose()Microsoft AI voices through Azure Speech's SSML REST API.
Args
url- Full synthesis POST endpoint, or MICROSOFT_AI_TTS_URL. Its host/path are used exactly as supplied and take precedence over region.
region- Public-cloud Azure Speech region, or MICROSOFT_AI_TTS_REGION. Used only when the URL is unset, not when it is empty or whitespace.
model- Voice model, or MICROSOFT_AI_TTS_MODEL, e.g. MAI-Voice-2-Flash. Appended to a short voice name; optional when voice is a full ID, which it must then match (case-insensitively), not an alias.
voice- Voice name such as en-US-Harper, or a full Azure Speech voice ID such as en-US-Harper:MAI-Voice-2-Flash, or MICROSOFT_AI_TTS_VOICE.
language- SSML language, default en-US.
sample_rate- Expected WAV sample rate, or MICROSOFT_AI_TTS_SAMPLE_RATE.
api_key- Azure Speech subscription key, or MICROSOFT_AI_TTS_API_KEY.
headers- Explicit headers instead of api_key/environment lookup.
http_session- Optional caller-owned aiohttp session.
env_file- Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor arguments override environment variables, which override this file.
request_timeout- Total request/body deadline in seconds.
max_text_length- Client-side character limit, not a claimed provider limit.
max_audio_bytes- Maximum complete WAV response size in bytes.
AgentSession wraps this nonstreaming provider in its existing sentence StreamAdapter. Native text/audio streaming and JSON/base64 responses are unsupported.
Ancestors
- livekit.agents.tts.tts.TTS
- abc.ABC
- EventEmitter
- typing.Generic
Instance variables
prop model : str-
Expand source code
@property def model(self) -> str: return self._opts.modelGet the model name/identifier for this TTS instance.
Returns
The model name if available, "unknown" otherwise.
Note
Plugins should override this property to provide their model information.
prop provider : str-
Expand source code
@property def provider(self) -> str: return "Microsoft AI"Get the provider name/identifier for this TTS instance.
Returns
The provider name if available, "unknown" otherwise.
Note
Plugins should override this property to provide their provider information.
Methods
async def aclose(self) ‑> None-
Expand source code
async def aclose(self) -> None: self._closed = True await asyncio.gather(*(stream.aclose() for stream in list(self._streams))) await self._client.aclose() def synthesize(self,
text: str,
*,
conn_options: APIConnectOptions = APIConnectOptions(max_retry=3, retry_interval=2.0, timeout=10.0)) ‑> livekit.plugins.microsoft_ai.tts.ChunkedStream-
Expand source code
def synthesize( self, text: str, *, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS ) -> ChunkedStream: if self._closed: raise RuntimeError("Microsoft AI TTS is closed") if not text.strip(): raise ValueError("Microsoft AI TTS requires nonempty text") if len(text) > self._max_text_length: raise ValueError("Microsoft AI TTS text exceeds max_text_length") if re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f\ud800-\udfff\ufffe\uffff]", text): raise ValueError("Microsoft AI TTS text contains characters invalid in XML") positive_timeout(conn_options.timeout, "conn_options.timeout") stream = ChunkedStream(tts=self, input_text=text, conn_options=conn_options) self._streams.add(stream) return stream
Inherited members