Module livekit.plugins.microsoft_ai

Microsoft AI STT and TTS using explicit, deployment-specific endpoints.

TTS uses Azure Speech; STT uses the transcription WebSocket API. See the package README for supported contracts and the limits of the bounded live smoke coverage.

Classes

class STT (*,
vad: vad.VAD | None,
url: str | None = None,
model: str | None = None,
api_key: str | None = None,
auth_header: "Literal['Authorization', 'api-key'] | None" = None,
headers: Mapping[str, str] | None = None,
language: str | None = None,
http_session: aiohttp.ClientSession | None = None,
env_file: str | Path | None = None,
max_buffered_audio: float = 5.0)
Expand source code
class STT(stt.STT):
    """Native streaming Microsoft AI transcription with explicit client commits.

    See the package README for deployment-specific contract and validation limits.

    Args:
        vad: A VAD emitting ordered INFERENCE_DONE events even during silence,
            with input-relative timestamps, and START_OF_SPEECH
            frames containing the detected onset/prefix through that timestamp
            (e.g. LiveKit's Silero VAD). Idle audio is withheld; these public frames
            recover the onset even when the start notification is delayed.
            Required explicitly: pass None only when driving flush/end_input yourself.
            AgentSession's separate VAD does not commit a native STT stream.
        url: Full WebSocket URL, or MICROSOFT_AI_STT_URL. No paths/query parameters
            are added automatically.
        model: Deployment model ID, or MICROSOFT_AI_STT_MODEL. No model is assumed.
        api_key: Credential, or MICROSOFT_AI_STT_API_KEY, sent using auth_header.
        auth_header: Authorization (the default, with Bearer prefix) or api-key
            (raw credential), or MICROSOFT_AI_STT_AUTH_HEADER. No auth fallback occurs.
        headers: Explicit authentication headers instead of api_key/auth_header
            environment lookup. Cannot be combined with either constructor argument.
        language: Optional transcription language hint, or MICROSOFT_AI_STT_LANGUAGE.
        http_session: Optional caller-owned aiohttp session.
        env_file: Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor
            arguments override environment variables, which override this file.
        max_buffered_audio: Client-side queued-audio and VAD start-prefix limits
            in seconds. Queue overflow or an oversized VAD prefix fails explicitly
            instead of silently dropping speech.
    """

    def __init__(
        self,
        *,
        vad: vad.VAD | None,
        url: str | None = None,
        model: str | None = None,
        api_key: str | None = None,
        auth_header: Literal["Authorization", "api-key"] | None = None,
        headers: Mapping[str, str] | None = None,
        language: str | None = None,
        http_session: aiohttp.ClientSession | None = None,
        env_file: str | Path | None = None,
        max_buffered_audio: float = 5.0,
    ) -> None:
        super().__init__(
            capabilities=stt.STTCapabilities(
                streaming=True, interim_results=True, offline_recognize=False
            )
        )
        config = Configuration(env_file)
        positive_timeout(max_buffered_audio, "max_buffered_audio")
        if language is None:
            language = config.get("MICROSOFT_AI_STT_LANGUAGE") or None
        if language is not None and not language.strip():
            raise ValueError("language must be nonempty when supplied")
        self._model = config.required(model, "MICROSOFT_AI_STT_MODEL")
        self._client = HTTPClient(
            config=config,
            service="STT",
            url=url,
            api_key=api_key,
            headers=headers,
            http_session=http_session,
            auth_header=auth_header,
        )
        self._vad = vad
        self._language = language
        self._max_buffered_audio = max_buffered_audio
        self._streams: weakref.WeakSet[SpeechStream] = weakref.WeakSet()
        self._closed = False

    @property
    def model(self) -> str:
        return self._model

    @property
    def provider(self) -> str:
        return "Microsoft AI"

    async def _recognize_impl(
        self,
        buffer: utils.AudioBuffer,
        *,
        language: NotGivenOr[str] = NOT_GIVEN,
        conn_options: APIConnectOptions,
    ) -> stt.SpeechEvent:
        raise NotImplementedError("Microsoft AI STT supports stream(), not batch recognize()")

    def stream(
        self,
        *,
        language: NotGivenOr[str] = NOT_GIVEN,
        conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS,
    ) -> SpeechStream:
        """Open a transcription stream; flush commits a turn, end_input also drains finals."""
        if self._closed:
            raise RuntimeError("Microsoft AI STT is closed")
        positive_timeout(conn_options.timeout, "conn_options.timeout")
        selected_language = language if utils.is_given(language) else self._language
        if selected_language is not None and not selected_language.strip():
            raise ValueError("language must be nonempty when supplied")
        stream = SpeechStream(stt=self, language=selected_language, conn_options=conn_options)
        self._streams.add(stream)
        return stream

    async def aclose(self) -> None:
        self._closed = True
        await asyncio.gather(*(stream.aclose() for stream in list(self._streams)))
        await self._client.aclose()

Native streaming Microsoft AI transcription with explicit client commits.

See the package README for deployment-specific contract and validation limits.

Args

vad
A VAD emitting ordered INFERENCE_DONE events even during silence, with input-relative timestamps, and START_OF_SPEECH frames containing the detected onset/prefix through that timestamp (e.g. LiveKit's Silero VAD). Idle audio is withheld; these public frames recover the onset even when the start notification is delayed. Required explicitly: pass None only when driving flush/end_input yourself. AgentSession's separate VAD does not commit a native STT stream.
url
Full WebSocket URL, or MICROSOFT_AI_STT_URL. No paths/query parameters are added automatically.
model
Deployment model ID, or MICROSOFT_AI_STT_MODEL. No model is assumed.
api_key
Credential, or MICROSOFT_AI_STT_API_KEY, sent using auth_header.
auth_header
Authorization (the default, with Bearer prefix) or api-key (raw credential), or MICROSOFT_AI_STT_AUTH_HEADER. No auth fallback occurs.
headers
Explicit authentication headers instead of api_key/auth_header environment lookup. Cannot be combined with either constructor argument.
language
Optional transcription language hint, or MICROSOFT_AI_STT_LANGUAGE.
http_session
Optional caller-owned aiohttp session.
env_file
Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor arguments override environment variables, which override this file.
max_buffered_audio
Client-side queued-audio and VAD start-prefix limits in seconds. Queue overflow or an oversized VAD prefix fails explicitly instead of silently dropping speech.

Ancestors

  • livekit.agents.stt.stt.STT
  • abc.ABC
  • EventEmitter
  • typing.Generic

Instance variables

prop model : str
Expand source code
@property
def model(self) -> str:
    return self._model

Get the model name/identifier for this STT instance.

Returns

The model name if available, "unknown" otherwise.

Note

Plugins should override this property to provide their model information.

prop provider : str
Expand source code
@property
def provider(self) -> str:
    return "Microsoft AI"

Get the provider name/identifier for this STT instance.

Returns

The provider name if available, "unknown" otherwise.

Note

Plugins should override this property to provide their provider information.

Methods

async def aclose(self) ‑> None
Expand source code
async def aclose(self) -> None:
    self._closed = True
    await asyncio.gather(*(stream.aclose() for stream in list(self._streams)))
    await self._client.aclose()

Close the STT, and every stream/requests associated with it

def stream(self,
*,
language: NotGivenOr[str] = NOT_GIVEN,
conn_options: APIConnectOptions = APIConnectOptions(max_retry=3, retry_interval=2.0, timeout=10.0)) ‑> livekit.plugins.microsoft_ai.stt.SpeechStream
Expand source code
def stream(
    self,
    *,
    language: NotGivenOr[str] = NOT_GIVEN,
    conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS,
) -> SpeechStream:
    """Open a transcription stream; flush commits a turn, end_input also drains finals."""
    if self._closed:
        raise RuntimeError("Microsoft AI STT is closed")
    positive_timeout(conn_options.timeout, "conn_options.timeout")
    selected_language = language if utils.is_given(language) else self._language
    if selected_language is not None and not selected_language.strip():
        raise ValueError("language must be nonempty when supplied")
    stream = SpeechStream(stt=self, language=selected_language, conn_options=conn_options)
    self._streams.add(stream)
    return stream

Open a transcription stream; flush commits a turn, end_input also drains finals.

Inherited members

class TTS (*,
url: str | None = None,
region: str | None = None,
model: str | None = None,
voice: str | None = None,
language: str = 'en-US',
sample_rate: int | None = None,
api_key: str | None = None,
headers: Mapping[str, str] | None = None,
http_session: aiohttp.ClientSession | None = None,
env_file: str | Path | None = None,
request_timeout: float = 30.0,
max_text_length: int = 4096,
max_audio_bytes: int = 10485760)
Expand source code
class TTS(tts.TTS):
    """Microsoft AI voices through Azure Speech's SSML REST API.

    Args:
        url: Full synthesis POST endpoint, or MICROSOFT_AI_TTS_URL. Its host/path
            are used exactly as supplied and take precedence over region.
        region: Public-cloud Azure Speech region, or MICROSOFT_AI_TTS_REGION.
            Used only when the URL is unset, not when it is empty or whitespace.
        model: Voice model, or MICROSOFT_AI_TTS_MODEL, e.g. MAI-Voice-2-Flash. Appended
            to a short voice name; optional when voice is a full ID, which it must
            then match (case-insensitively), not an alias.
        voice: Voice name such as en-US-Harper, or a full Azure Speech voice ID such
            as en-US-Harper:MAI-Voice-2-Flash, or MICROSOFT_AI_TTS_VOICE.
        language: SSML language, default en-US.
        sample_rate: Expected WAV sample rate, or MICROSOFT_AI_TTS_SAMPLE_RATE.
        api_key: Azure Speech subscription key, or MICROSOFT_AI_TTS_API_KEY.
        headers: Explicit headers instead of api_key/environment lookup.
        http_session: Optional caller-owned aiohttp session.
        env_file: Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor
            arguments override environment variables, which override this file.
        request_timeout: Total request/body deadline in seconds.
        max_text_length: Client-side character limit, not a claimed provider limit.
        max_audio_bytes: Maximum complete WAV response size in bytes.

    AgentSession wraps this nonstreaming provider in its existing sentence
    StreamAdapter. Native text/audio streaming and JSON/base64 responses are unsupported.
    """

    def __init__(
        self,
        *,
        url: str | None = None,
        region: str | None = None,
        model: str | None = None,
        voice: str | None = None,
        language: str = "en-US",
        sample_rate: int | None = None,
        api_key: str | None = None,
        headers: Mapping[str, str] | None = None,
        http_session: aiohttp.ClientSession | None = None,
        env_file: str | Path | None = None,
        request_timeout: float = 30.0,
        max_text_length: int = 4096,
        max_audio_bytes: int = 10 * 1024 * 1024,
    ) -> None:
        config = Configuration(env_file)
        if sample_rate is None:
            try:
                sample_rate = int(config.get("MICROSOFT_AI_TTS_SAMPLE_RATE") or "")
            except ValueError:
                raise ValueError("Set MICROSOFT_AI_TTS_SAMPLE_RATE or pass sample_rate") from None
        if (
            isinstance(sample_rate, bool)
            or not isinstance(sample_rate, int)
            or sample_rate not in _WAV_FORMATS
        ):
            raise ValueError("sample_rate must be one of 8000, 22050, 24000, 44100, 48000")
        for name, value in (
            ("max_text_length", max_text_length),
            ("max_audio_bytes", max_audio_bytes),
        ):
            if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
                raise ValueError(f"{name} must be a positive integer")
        positive_timeout(request_timeout, "request_timeout")
        super().__init__(
            capabilities=tts.TTSCapabilities(streaming=False),
            sample_rate=sample_rate,
            num_channels=1,
        )
        voice = config.required(voice, "MICROSOFT_AI_TTS_VOICE")
        voice_name, separator, voice_model = voice.rpartition(":")
        if separator:
            if not voice_name or not voice_model:
                raise ValueError("voice must be a voice name or a full voice ID such as name:model")
            if model is None:
                model = config.get("MICROSOFT_AI_TTS_MODEL")
                if model is None:
                    model = voice_model
            model = config.required(model, "MICROSOFT_AI_TTS_MODEL")
            if voice_model.casefold() != model.casefold():
                raise ValueError("full voice ID must end in the configured model name")
        else:
            model = config.required(model, "MICROSOFT_AI_TTS_MODEL")
            voice = f"{voice}:{model}"
        if re.fullmatch(r"[A-Za-z]{2,3}(?:-[A-Za-z0-9]{2,8})*", language) is None:
            raise ValueError("language must be a language tag such as en-US")
        self._opts = _TTSOptions(
            model=model, voice=voice, sample_rate=sample_rate, language=language
        )
        self._client = HTTPClient(
            config=config,
            service="TTS",
            url=_endpoint(config, url, region),
            api_key=api_key,
            headers=headers,
            http_session=http_session,
        )
        self._client.headers = {
            key: value
            for key, value in self._client.headers.items()
            if key.lower() not in ("accept", "content-type", "x-microsoft-outputformat")
        }
        self._client.headers.update(
            {
                "Accept": "audio/wav",
                "Content-Type": "application/ssml+xml",
                "X-Microsoft-OutputFormat": _WAV_FORMATS[sample_rate],
            }
        )
        self._request_timeout = request_timeout
        self._max_text_length = max_text_length
        self._max_audio_bytes = max_audio_bytes
        self._streams: weakref.WeakSet[ChunkedStream] = weakref.WeakSet()
        self._closed = False

    @property
    def model(self) -> str:
        return self._opts.model

    @property
    def provider(self) -> str:
        return "Microsoft AI"

    def synthesize(
        self, text: str, *, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS
    ) -> ChunkedStream:
        if self._closed:
            raise RuntimeError("Microsoft AI TTS is closed")
        if not text.strip():
            raise ValueError("Microsoft AI TTS requires nonempty text")
        if len(text) > self._max_text_length:
            raise ValueError("Microsoft AI TTS text exceeds max_text_length")
        if re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f\ud800-\udfff\ufffe\uffff]", text):
            raise ValueError("Microsoft AI TTS text contains characters invalid in XML")
        positive_timeout(conn_options.timeout, "conn_options.timeout")
        stream = ChunkedStream(tts=self, input_text=text, conn_options=conn_options)
        self._streams.add(stream)
        return stream

    async def aclose(self) -> None:
        self._closed = True
        await asyncio.gather(*(stream.aclose() for stream in list(self._streams)))
        await self._client.aclose()

Microsoft AI voices through Azure Speech's SSML REST API.

Args

url
Full synthesis POST endpoint, or MICROSOFT_AI_TTS_URL. Its host/path are used exactly as supplied and take precedence over region.
region
Public-cloud Azure Speech region, or MICROSOFT_AI_TTS_REGION. Used only when the URL is unset, not when it is empty or whitespace.
model
Voice model, or MICROSOFT_AI_TTS_MODEL, e.g. MAI-Voice-2-Flash. Appended to a short voice name; optional when voice is a full ID, which it must then match (case-insensitively), not an alias.
voice
Voice name such as en-US-Harper, or a full Azure Speech voice ID such as en-US-Harper:MAI-Voice-2-Flash, or MICROSOFT_AI_TTS_VOICE.
language
SSML language, default en-US.
sample_rate
Expected WAV sample rate, or MICROSOFT_AI_TTS_SAMPLE_RATE.
api_key
Azure Speech subscription key, or MICROSOFT_AI_TTS_API_KEY.
headers
Explicit headers instead of api_key/environment lookup.
http_session
Optional caller-owned aiohttp session.
env_file
Explicit dotenv file, or MICROSOFT_AI_ENV_FILE. Constructor arguments override environment variables, which override this file.
request_timeout
Total request/body deadline in seconds.
max_text_length
Client-side character limit, not a claimed provider limit.
max_audio_bytes
Maximum complete WAV response size in bytes.

AgentSession wraps this nonstreaming provider in its existing sentence StreamAdapter. Native text/audio streaming and JSON/base64 responses are unsupported.

Ancestors

  • livekit.agents.tts.tts.TTS
  • abc.ABC
  • EventEmitter
  • typing.Generic

Instance variables

prop model : str
Expand source code
@property
def model(self) -> str:
    return self._opts.model

Get the model name/identifier for this TTS instance.

Returns

The model name if available, "unknown" otherwise.

Note

Plugins should override this property to provide their model information.

prop provider : str
Expand source code
@property
def provider(self) -> str:
    return "Microsoft AI"

Get the provider name/identifier for this TTS instance.

Returns

The provider name if available, "unknown" otherwise.

Note

Plugins should override this property to provide their provider information.

Methods

async def aclose(self) ‑> None
Expand source code
async def aclose(self) -> None:
    self._closed = True
    await asyncio.gather(*(stream.aclose() for stream in list(self._streams)))
    await self._client.aclose()
def synthesize(self,
text: str,
*,
conn_options: APIConnectOptions = APIConnectOptions(max_retry=3, retry_interval=2.0, timeout=10.0)) ‑> livekit.plugins.microsoft_ai.tts.ChunkedStream
Expand source code
def synthesize(
    self, text: str, *, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS
) -> ChunkedStream:
    if self._closed:
        raise RuntimeError("Microsoft AI TTS is closed")
    if not text.strip():
        raise ValueError("Microsoft AI TTS requires nonempty text")
    if len(text) > self._max_text_length:
        raise ValueError("Microsoft AI TTS text exceeds max_text_length")
    if re.search(r"[\x00-\x08\x0b\x0c\x0e-\x1f\ud800-\udfff\ufffe\uffff]", text):
        raise ValueError("Microsoft AI TTS text contains characters invalid in XML")
    positive_timeout(conn_options.timeout, "conn_options.timeout")
    stream = ChunkedStream(tts=self, input_text=text, conn_options=conn_options)
    self._streams.add(stream)
    return stream

Inherited members