Module livekit.plugins.azure.realtime.utils
Functions
def azure_item_to_livekit_item(item: ResponseItem) ‑> livekit.agents.llm.chat_context.ChatMessage | livekit.agents.llm.chat_context.FunctionCall | livekit.agents.llm.chat_context.FunctionCallOutput | livekit.agents.llm.chat_context.AgentHandoff | livekit.agents.llm.chat_context.AgentConfigUpdate-
Expand source code
def azure_item_to_livekit_item(item: ResponseItem) -> llm.ChatItem: """Convert a conversation item created by Azure Voice Live to a LiveKit chat item.""" if not item.id: raise ValueError("conversation item has no id") if isinstance(item, ResponseFunctionCallItem): return llm.FunctionCall( id=item.id, call_id=item.call_id, name=item.name, arguments=item.arguments or "", ) if isinstance(item, ResponseFunctionCallOutputItem): return llm.FunctionCallOutput( id=item.id, call_id=item.call_id, output=item.output, is_error=False, ) if isinstance(item, ResponseMessageItem): raw_role = item.role.value if isinstance(item.role, Enum) else item.role role = _CHAT_ROLES.get(raw_role) if role is None: raise ValueError(f"Unsupported role: {raw_role}") content: list[llm.ChatContent] = [] for part in item.content or []: # text parts carry `text`, audio parts carry the `transcript` of the audio text = getattr(part, "text", None) or getattr(part, "transcript", None) if isinstance(text, str) and text: content.append(text) return llm.ChatMessage(id=item.id, role=role, content=content) raise ValueError(f"Unsupported item type: {item.type}")Convert a conversation item created by Azure Voice Live to a LiveKit chat item.
def livekit_item_to_azure_item(item: llm.ChatItem) ‑> azure.ai.voicelive.models._models.SystemMessageItem | azure.ai.voicelive.models._models.UserMessageItem | azure.ai.voicelive.models._models.AssistantMessageItem | azure.ai.voicelive.models._models.FunctionCallItem | azure.ai.voicelive.models._models.FunctionCallOutputItem-
Expand source code
def livekit_item_to_azure_item(item: llm.ChatItem) -> AzureConversationItem: if item.type == "function_call_output": return FunctionCallOutputItem(call_id=item.call_id, output=item.output, id=item.id) if item.type == "function_call": return FunctionCallItem( call_id=item.call_id, name=item.name, arguments=item.arguments, id=item.id, ) if item.type == "message": if item.role in ("system", "developer"): content_parts: list[MessageContentPart] = [ InputTextContentPart(text=c) for c in item.content if isinstance(c, str) ] return SystemMessageItem(content=content_parts, id=item.id) if item.role == "assistant": content_parts = [ OutputTextContentPart(text=c) for c in item.content if isinstance(c, str) ] return AssistantMessageItem(content=content_parts, id=item.id) if item.role == "user": content_parts = [] for c in item.content: if isinstance(c, str): content_parts.append(InputTextContentPart(text=c)) elif isinstance(c, llm.AudioContent): encoded_audio = base64.b64encode(rtc.combine_audio_frames(c.frame).data).decode( "utf-8" ) content_parts.append( InputAudioContentPart(audio=encoded_audio, transcript=c.transcript) ) return UserMessageItem(content=content_parts, id=item.id) raise ValueError(f"Unsupported role: {item.role}") raise ValueError(f"Unsupported item type: {item.type}") def livekit_tool_to_azure_tool(tool: llm.Tool) ‑> azure.ai.voicelive.models._models.FunctionTool | None-
Expand source code
def livekit_tool_to_azure_tool(tool: llm.Tool) -> FunctionTool | None: """Convert LiveKit Tool to Azure FunctionTool format. Returns None for unsupported tool types (e.g. ProviderTool). """ from livekit.agents.llm import utils as llm_utils if isinstance(tool, llm.FunctionTool): schema = llm_utils.build_legacy_openai_schema(tool, internally_tagged=True) return FunctionTool( name=schema["name"], description=schema.get("description", ""), parameters=schema.get("parameters", {}), ) if isinstance(tool, llm.RawFunctionTool): raw_schema = tool.info.raw_schema return FunctionTool( name=tool.info.name, description=raw_schema.get("description", ""), parameters=raw_schema.get("parameters", {}), ) logger.warning( "Azure Voice Live doesn't support this tool type, skipping it", extra={"tool_type": type(tool).__name__}, ) return NoneConvert LiveKit Tool to Azure FunctionTool format.
Returns None for unsupported tool types (e.g. ProviderTool).
def livekit_tools_to_azure_tools(tools: Sequence[llm.Tool]) ‑> list[azure.ai.voicelive.models._models.Tool]-
Expand source code
def livekit_tools_to_azure_tools(tools: Sequence[llm.Tool]) -> list[Tool]: """Convert LiveKit tools to Azure function tools, skipping unsupported tool types.""" azure_tools: list[Tool] = [] for tool in tools: if (azure_tool := livekit_tool_to_azure_tool(tool)) is not None: azure_tools.append(azure_tool) return azure_toolsConvert LiveKit tools to Azure function tools, skipping unsupported tool types.
def to_audio_transcription(audio_transcription: NotGivenOr[AudioInputTranscriptionOptions | None],
*,
model: str) ‑> azure.ai.voicelive.models._models.AudioInputTranscriptionOptions | None-
Expand source code
def to_audio_transcription( audio_transcription: NotGivenOr[AudioInputTranscriptionOptions | None], *, model: str, ) -> AudioInputTranscriptionOptions | None: """Convert audio transcription configuration to Azure AudioInputTranscriptionOptions format. Args: audio_transcription: Audio transcription options. If NOT_GIVEN, returns the default config of the model, azure-speech or whisper-1 (see `uses_azure_speech`). If None, transcription is disabled. Otherwise, returns the provided config. model: The Voice Live model of the session. Returns: AudioInputTranscriptionOptions or None if transcription is disabled. """ if not is_given(audio_transcription): if uses_azure_speech(model): return AZURE_SPEECH_INPUT_AUDIO_TRANSCRIPTION return DEFAULT_INPUT_AUDIO_TRANSCRIPTION if audio_transcription is None: return None return audio_transcriptionConvert audio transcription configuration to Azure AudioInputTranscriptionOptions format.
Args
audio_transcription- Audio transcription options. If NOT_GIVEN, returns the default config
of the model, azure-speech or whisper-1 (see
uses_azure_speech()). If None, transcription is disabled. Otherwise, returns the provided config. model- The Voice Live model of the session.
Returns
AudioInputTranscriptionOptions or None if transcription is disabled.
def to_azure_response_tool_choice(tool_choice: llm.ToolChoice | None) ‑> str-
Expand source code
def to_azure_response_tool_choice(tool_choice: llm.ToolChoice | None) -> str: """Convert a LiveKit ToolChoice to the tool_choice of a single response. The response-level field is a string: a mode (``auto``, ``none``, ``required``) or the name of the function the model must call. """ if isinstance(tool_choice, str): return tool_choice if isinstance(tool_choice, dict) and tool_choice.get("type") == "function": return tool_choice["function"]["name"] return ToolChoiceLiteral.AUTO.valueConvert a LiveKit ToolChoice to the tool_choice of a single response.
The response-level field is a string: a mode (
auto,none,required) or the name of the function the model must call. def to_azure_tool_choice(tool_choice: llm.ToolChoice | None) ‑> azure.ai.voicelive.models._enums.ToolChoiceLiteral | azure.ai.voicelive.models._models.ToolChoiceSelection-
Expand source code
def to_azure_tool_choice( tool_choice: llm.ToolChoice | None, ) -> ToolChoiceLiteral | ToolChoiceSelection: """Convert a LiveKit ToolChoice to Azure's session-level tool_choice format.""" if isinstance(tool_choice, str): return ToolChoiceLiteral(tool_choice) if isinstance(tool_choice, dict) and tool_choice.get("type") == "function": return ToolChoiceFunctionSelection(name=tool_choice["function"]["name"]) return DEFAULT_TOOL_CHOICEConvert a LiveKit ToolChoice to Azure's session-level tool_choice format.
def to_turn_detection(turn_detection: NotGivenOr[TurnDetection | None]) ‑> azure.ai.voicelive.models._models.TurnDetection | None-
Expand source code
def to_turn_detection( turn_detection: NotGivenOr[TurnDetection | None], ) -> TurnDetection | None: """Convert turn detection configuration to Azure TurnDetection format. Accepts any TurnDetection subclass including: - ServerVad: Basic server-side VAD - AzureSemanticVad: Semantic VAD (multilingual) - AzureSemanticVadEn: English-only semantic VAD - AzureSemanticVadMultilingual: Explicit multilingual semantic VAD """ if not is_given(turn_detection): return DEFAULT_TURN_DETECTION if turn_detection is None: return None return turn_detectionConvert turn detection configuration to Azure TurnDetection format.
Accepts any TurnDetection subclass including: - ServerVad: Basic server-side VAD - AzureSemanticVad: Semantic VAD (multilingual) - AzureSemanticVadEn: English-only semantic VAD - AzureSemanticVadMultilingual: Explicit multilingual semantic VAD
def uses_azure_speech(model: str) ‑> bool-
Expand source code
def uses_azure_speech(model: str) -> bool: """Whether the input audio of the model is transcribed with azure-speech, not whisper-1. Voice Live documents whisper-1 for gpt-realtime and gpt-realtime-mini, and azure-speech for the non-multimodal (text) models and phi4-mm-realtime, e.g. gpt-4.1 answers whisper-1 with invalid_input_audio_transcription_model. The multimodal models have "realtime" in their name, those without documented transcription models (e.g. gpt-realtime-1.5, azure-realtime) keep whisper-1. See https://learn.microsoft.com/azure/ai-services/speech-service/voice-live-how-to and https://learn.microsoft.com/azure/ai-services/speech-service/voice-live-language-support """ name = model.lower() return "realtime" not in name or name.startswith("phi")Whether the input audio of the model is transcribed with azure-speech, not whisper-1.
Voice Live documents whisper-1 for gpt-realtime and gpt-realtime-mini, and azure-speech for the non-multimodal (text) models and phi4-mm-realtime, e.g. gpt-4.1 answers whisper-1 with invalid_input_audio_transcription_model. The multimodal models have "realtime" in their name, those without documented transcription models (e.g. gpt-realtime-1.5, azure-realtime) keep whisper-1.
See https://learn.microsoft.com/azure/ai-services/speech-service/voice-live-how-to and https://learn.microsoft.com/azure/ai-services/speech-service/voice-live-language-support