steep-balloon-41261
07/30/2025, 5:49 PMtall-airline-16756
07/30/2025, 5:49 PMtall-airline-16756
07/30/2025, 5:53 PMpush_to_talk.py example.
• Functionality to switch between these agents while preserving chat context.
Code:
import logging
from dataclasses import dataclass
from typing import Optional
from dotenv import load_dotenv
from livekit import rtc
from livekit.agents import (
Agent,
AgentSession,
ChatContext,
JobContext,
JobRequest,
RoomIO,
RunContext,
WorkerOptions,
cli,
)
from livekit.agents.llm import function_tool, ChatMessage, StopResponse
from livekit.plugins import deepgram, openai, silero, cartesia
logger = logging.getLogger("multi-agent")
logger.setLevel(<http://logging.INFO|logging.INFO>)
load_dotenv()
@dataclass
class SharedData:
"""Shared data structure for passing information between agents"""
user_name: Optional[str] = None
conversation_context: Optional[str] = None
class StarterAgent(Agent):
"""Simple starter agent with all needed features from plugins"""
def __init__(self) -> None:
super().__init__(
instructions="""You are a helpful voice assistant built by LiveKit.
You can help users with general questions and tasks.
When a user wants to use push-to-talk functionality,
you can switch them to the push-to-talk agent.""",
stt=deepgram.STT(model="nova-3"),
llm=openai.LLM(model="gpt-4o-mini"),
tts=openai.TTS(voice="echo"),
vad=silero.VAD.load(),
)
async def on_enter(self):
self.session.generate_reply(
instructions="Greet the user and explain you're the starter agent. "
"Mention that you can switch to push-to-talk mode if needed."
)
@function_tool
async def switch_to_push_to_talk(self, context: RunContext[SharedData]):
context.userdata.conversation_context = "Switched from starter agent"
ptt_agent = PushToTalkAgent(chat_ctx=context.chat_ctx)
<http://logger.info|logger.info>("Switching to push-to-talk agent")
return ptt_agent, "Switching to push-to-talk mode. You'll need to press and hold to speak."
class PushToTalkAgent(Agent):
"""Agent based on push_to_talk.py example"""
def __init__(self, *, chat_ctx: Optional[ChatContext] = None) -> None:
super().__init__(
instructions="""You are a push-to-talk voice assistant.
Users must press and hold a button to speak to you.
You can switch back to the regular starter agent if requested.""",
stt=deepgram.STT(),
llm=openai.LLM(model="gpt-4o-mini"),
tts=cartesia.TTS(),
chat_ctx=chat_ctx,
)
async def on_enter(self):
self.session.input.set_audio_enabled(False)
self.session.generate_reply(
instructions="Explain that you're now in push-to-talk mode and "
"the user needs to press and hold to speak."
)
@function_tool
async def switch_to_starter_agent(self, context: RunContext[SharedData]):
context.userdata.conversation_context = "Switched from push-to-talk agent"
starter_agent = StarterAgent()
<http://logger.info|logger.info>("Switching back to starter agent")
return starter_agent, "Switching back to regular voice mode."
async def entrypoint(ctx: JobContext):
await ctx.connect()
session = AgentSession[SharedData](
vad=silero.VAD.load(),
stt=deepgram.STT(model="nova-3"),
llm=openai.LLM(model="gpt-4o-mini"),
tts=openai.TTS(voice="echo"),
userdata=SharedData(),
turn_detection="manual",
)
room_io = RoomIO(session, room=ctx.room)
await room_io.start()
@ctx.room.local_participant.register_rpc_method("start_turn")
async def start_turn(data: rtc.RpcInvocationData):
session.interrupt()
session.clear_user_turn()
room_io.set_participant(data.caller_identity)
session.input.set_audio_enabled(True)
<http://logger.info|logger.info>("Push-to-talk started")
@ctx.room.local_participant.register_rpc_method("end_turn")
async def end_turn(data: rtc.RpcInvocationData):
session.input.set_audio_enabled(False)
session.commit_user_turn(transcript_timeout=10.0)
<http://logger.info|logger.info>("Push-to-talk ended")
await session.start(agent=StarterAgent(), room=ctx.room)
if __name__ == "__main__":
cli.run_app(WorkerOptions(entrypoint_fnc=entrypoint))
This implementation includes:
• Agent switching with preserved chat context.
• Push-to-talk functionality with manual turn detection.
• Integration with plugins like STT, LLM, TTS, and VAD.
Let me know if you need further clarification or additional features!
Sources: examples/voice_agents/push_to_talk.py | README.md