This message was deleted.
# ask-ai
s
This message was deleted.
t
Thanks for reaching out, Ibrahem. I'm an AI assistant and I'm here to help. Give me a moment while I research this topic.
Here is a Python implementation for a multi-agent system that includes: • A Starter Agent with all necessary features from plugins (STT, LLM, TTS, VAD). • A Push-to-Talk Agent based on the
push_to_talk.py
example. • Functionality to switch between these agents while preserving chat context. Code:
Copy code
import logging
from dataclasses import dataclass
from typing import Optional

from dotenv import load_dotenv
from livekit import rtc
from livekit.agents import (
    Agent,
    AgentSession,
    ChatContext,
    JobContext,
    JobRequest,
    RoomIO,
    RunContext,
    WorkerOptions,
    cli,
)
from livekit.agents.llm import function_tool, ChatMessage, StopResponse
from livekit.plugins import deepgram, openai, silero, cartesia

logger = logging.getLogger("multi-agent")
logger.setLevel(<http://logging.INFO|logging.INFO>)
load_dotenv()

@dataclass
class SharedData:
    """Shared data structure for passing information between agents"""
    user_name: Optional[str] = None
    conversation_context: Optional[str] = None

class StarterAgent(Agent):
    """Simple starter agent with all needed features from plugins"""
    def __init__(self) -> None:
        super().__init__(
            instructions="""You are a helpful voice assistant built by LiveKit. 
            You can help users with general questions and tasks. 
            When a user wants to use push-to-talk functionality, 
            you can switch them to the push-to-talk agent.""",
            stt=deepgram.STT(model="nova-3"),
            llm=openai.LLM(model="gpt-4o-mini"),
            tts=openai.TTS(voice="echo"),
            vad=silero.VAD.load(),
        )

    async def on_enter(self):
        self.session.generate_reply(
            instructions="Greet the user and explain you're the starter agent. "
            "Mention that you can switch to push-to-talk mode if needed."
        )

    @function_tool
    async def switch_to_push_to_talk(self, context: RunContext[SharedData]):
        context.userdata.conversation_context = "Switched from starter agent"
        ptt_agent = PushToTalkAgent(chat_ctx=context.chat_ctx)
        <http://logger.info|logger.info>("Switching to push-to-talk agent")
        return ptt_agent, "Switching to push-to-talk mode. You'll need to press and hold to speak."

class PushToTalkAgent(Agent):
    """Agent based on push_to_talk.py example"""
    def __init__(self, *, chat_ctx: Optional[ChatContext] = None) -> None:
        super().__init__(
            instructions="""You are a push-to-talk voice assistant. 
            Users must press and hold a button to speak to you. 
            You can switch back to the regular starter agent if requested.""",
            stt=deepgram.STT(),
            llm=openai.LLM(model="gpt-4o-mini"),
            tts=cartesia.TTS(),
            chat_ctx=chat_ctx,
        )

    async def on_enter(self):
        self.session.input.set_audio_enabled(False)
        self.session.generate_reply(
            instructions="Explain that you're now in push-to-talk mode and "
            "the user needs to press and hold to speak."
        )

    @function_tool
    async def switch_to_starter_agent(self, context: RunContext[SharedData]):
        context.userdata.conversation_context = "Switched from push-to-talk agent"
        starter_agent = StarterAgent()
        <http://logger.info|logger.info>("Switching back to starter agent")
        return starter_agent, "Switching back to regular voice mode."

async def entrypoint(ctx: JobContext):
    await ctx.connect()
    session = AgentSession[SharedData](
        vad=silero.VAD.load(),
        stt=deepgram.STT(model="nova-3"),
        llm=openai.LLM(model="gpt-4o-mini"),
        tts=openai.TTS(voice="echo"),
        userdata=SharedData(),
        turn_detection="manual",
    )
    room_io = RoomIO(session, room=ctx.room)
    await room_io.start()

    @ctx.room.local_participant.register_rpc_method("start_turn")
    async def start_turn(data: rtc.RpcInvocationData):
        session.interrupt()
        session.clear_user_turn()
        room_io.set_participant(data.caller_identity)
        session.input.set_audio_enabled(True)
        <http://logger.info|logger.info>("Push-to-talk started")

    @ctx.room.local_participant.register_rpc_method("end_turn")
    async def end_turn(data: rtc.RpcInvocationData):
        session.input.set_audio_enabled(False)
        session.commit_user_turn(transcript_timeout=10.0)
        <http://logger.info|logger.info>("Push-to-talk ended")

    await session.start(agent=StarterAgent(), room=ctx.room)

if __name__ == "__main__":
    cli.run_app(WorkerOptions(entrypoint_fnc=entrypoint))
This implementation includes: • Agent switching with preserved chat context. • Push-to-talk functionality with manual turn detection. • Integration with plugins like STT, LLM, TTS, and VAD. Let me know if you need further clarification or additional features! Sources: examples/voice_agents/push_to_talk.py | README.md