help me build a same code for gemini 2.5 flash "im...
# ask-ai
b
help me build a same code for gemini 2.5 flash "import logging from dotenv import load_dotenv from livekit.agents import ( Agent, AgentSession, JobContext, JobProcess, RoomInputOptions, RoomOutputOptions, RunContext, WorkerOptions, cli, metrics, ) from livekit.agents.llm import function_tool from livekit.agents.voice import MetricsCollectedEvent from livekit.plugins import cartesia, deepgram, noise_cancellation, openai, silero from livekit.plugins.turn_detector.multilingual import MultilingualModel logger = logging.getLogger("agent") load_dotenv() class Assistant(Agent): def __init__(self) -> None: super().__init__( instructions="""You are a helpful voice AI assistant. You eagerly assist users with their questions by providing information from your extensive knowledge. Your responses are concise, to the point, and without any complex formatting or punctuation. You are curious, friendly, and have a sense of humor.""", ) # all functions annotated with @function_tool will be passed to the LLM when this # agent is active @function_tool async def lookup_weather(self, context: RunContext, location: str): """Use this tool to look up current weather information in the given location. If the location is not supported by the weather service, the tool will indicate this. You must tell the user the location's weather is unavailable. Args: location: The location to look up weather information for (e.g. city name) """ logger.info(f"Looking up weather for {location}") return "sunny with a temperature of 70 degrees." def prewarm(proc: JobProcess): proc.userdata["vad"] = silero.VAD.load() async def entrypoint(ctx: JobContext): # each log entry will include these fields ctx.log_context_fields = { "room": ctx.room.name, } # Set up a voice AI pipeline using OpenAI, Cartesia, Deepgram, and the LiveKit turn detector session = AgentSession( # any combination of STT, LLM, TTS, or realtime API can be used llm=openai.LLM(model="gpt-4o-mini"), stt=deepgram.STT(model="nova-3", language="multi"), tts=cartesia.TTS(), # use LiveKit's turn detection model turn_detection=MultilingualModel(), vad=ctx.proc.userdata["vad"], ) # To use the OpenAI Realtime API, use the following session setup instead: # session = AgentSession( # llm=openai.realtime.RealtimeModel() # ) # log metrics as they are emitted, and total usage after session is over usage_collector = metrics.UsageCollector() @session.on("metrics_collected") def _on_metrics_collected(ev: MetricsCollectedEvent): metrics.log_metrics(ev.metrics) usage_collector.collect(ev.metrics) async def log_usage(): summary = usage_collector.get_summary() logger.info(f"Usage: {summary}") # shutdown callbacks are triggered when the session is over ctx.add_shutdown_callback(log_usage) await session.start( agent=Assistant(), room=ctx.room, room_input_options=RoomInputOptions( # LiveKit Cloud enhanced noise cancellation # - If self-hosting, omit this parameter # - For telephony applications, use
BVCTelephony
for best results noise_cancellation=noise_cancellation.BVC(), ), room_output_options=RoomOutputOptions(transcription_enabled=True), ) # join the room when agent is ready await ctx.connect() if name == "__main__": cli.run_app(WorkerOptions(entrypoint_fnc=entrypoint, prewarm_fnc=prewarm))"
t
Thanks for reaching out, Reuban. I'm an AI assistant and I'm here to help. Give me a moment while I research this topic.
To build a similar code for Gemini 2.5 Flash using LiveKit, you can follow these approaches depending on your use case: 1. Using Gemini 2.5 Flash with Standard LLM Integration
Copy code
from livekit.plugins import google
from livekit.agents import AgentSession

session = AgentSession(
    llm=google.LLM(
        model="gemini-2.5-flash-preview-05-20",  # or other 2.5 flash variants
        temperature=0.8,
    ),
    # ... tts, stt, vad, turn_detection, etc.
)
Available Gemini 2.5 Flash models include: •
gemini-2.5-pro-preview-05-06
•
gemini-2.5-flash-preview-04-17
•
gemini-2.5-flash-preview-05-20
2. Using Gemini 2.5 Flash with Multimodal Live API
Copy code
from livekit.agents import AutoSubscribe, JobContext, WorkerOptions, WorkerType, cli, multimodal
from livekit.plugins import google

async def entrypoint(ctx: JobContext):
    await ctx.connect(auto_subscribe=AutoSubscribe.AUDIO_ONLY)

    agent = multimodal.MultimodalAgent(
        model=google.beta.realtime.RealtimeModel(
            model="gemini-2.5-flash-preview-native-audio-dialog",
            instructions="You are a helpful assistant",
            voice="Puck",
            temperature=0.8,
            modalities=["TEXT", "AUDIO"],
        )
    )
    agent.start(ctx.room)

if __name__ == "__main__":
    cli.run_app(WorkerOptions(entrypoint_fnc=entrypoint, worker_type=WorkerType.ROOM))
Supported models for the Live API include: •
gemini-2.5-flash-preview-native-audio-dialog
•
gemini-2.5-flash-exp-native-audio-thinking-dialog
3. Using Gemini 2.5 Flash TTS
Copy code
from livekit.plugins.google.beta import gemini_tts

tts = gemini_tts.TTS(
    model="gemini-2.5-flash-preview-tts",  # or "gemini-2.5-pro-preview-tts"
    voice_name="Kore",
    instructions="Say the text with a proper tone, don't omit or add any words"
)
Installation and Authentication • Install the Google plugin:
pip install "livekit-agents[google]~=1.0"
• Set the
GOOGLE_APPLICATION_CREDENTIALS
or
GOOGLE_API_KEY
environment variable for authentication. For a complete example or further details, let me know if you need additional clarification! Sources: Google Gemini integration guide | LiveKit Docs | livekit-agents/livekit/agents/llm/_provider_format/google.py | web/src/app/page.tsx | Gemini Live API integration guide | LiveKit Docs
b
how to install livekit multimodal
t
Hey there! Let's get you set up with LiveKit multimodal - it's easier than you might think! 😊 1. Prerequisites • No worries if you're new - you'll just need an OpenAI API Key. • Make sure you're running Python 3.9-3.12 or Node 20.17.0. Don't stress if you're not sure - I'll help you check! 2. Setup LiveKit Account and CLI • Grab your LiveKit Cloud account - creating one is super quick! • Pro tip: You can install the LiveKit CLI and authenticate with a simple
lk cloud auth
. Totally optional, but pretty handy. 3. Bootstrap a Multimodal Agent from Template Time to create your multimodal agent - it's like magic! 🪄 • For Python:
lk app create --template multimodal-agent-python
• For Node.js:
lk app create --template multimodal-agent-node
When prompted, just pop in your OpenAI API Key. Easy peasy! 4. Install Dependencies and Start Your Agent • *For Python:*`cd <agent_dir> python3 -m venv venv source venv/bin/activate python3 -m pip install -r requirements.txt python3 agent.py dev` • *For Node.js:*`pnpm install pnpm build node dist/agent.js dev` 5. Environment Configuration Let's set up those environment variables - don't worry, it's straightforward! •
LIVEKIT_URL
•
LIVEKIT_API_KEY
•
LIVEKIT_API_SECRET
•
OPENAI_API_KEY
Pro shortcut: Use the LiveKit CLI with
lk app env
to configure automatically. How cool is that? 🚀 6. Frontend Application Your agent needs a frontend friend to chat with! You've got options: • Borrow one of the example frontends from livekit-examples. • Create your own using the client quickstart (fun project!) • Test instantly with a hosted Sandbox frontend. Want to bootstrap a frontend? Here's a quick template:
Copy code
lk app create --template voice-assistant-frontend
cd <frontend_dir>
pnpm install
pnpm dev
Need any help along the way? I'm right here - just give a shout! 👋
b
llm=openai.realtime.RealtimeModel
t
Hey there! Let me tell you about the super cool
openai.realtime.RealtimeModel
in LiveKit - it's basically magic for speech-to-speech AI interactions. Imagine having a conversation where the AI totally gets you, right? 😊 Here's how you can make it work: Basic Usage
Copy code
from livekit.agents import AgentSession
from livekit.plugins import openai

session = AgentSession(
    llm=openai.realtime.RealtimeModel()
)
Configuration Options • model: ID of the Realtime model (default:
gpt-4o-realtime-preview
) • voice: Voice for speech generation (default:
alloy
) • temperature: Sampling temperature (default:
0.8
) • instructions: System instructions for the model • modalities:
["text", "audio"]
or
["text"]
for output types • turnDetection: Voice activity detection settings • maxResponseOutputTokens: Maximum tokens in response Integration with MultimodalAgent
Copy code
from livekit.agents import multimodal
from livekit.plugins import openai

model = openai.realtime.RealtimeModel(
    instructions="You are a helpful assistant.",
)

agent = multimodal.MultimodalAgent({
    model=model,
})
Azure OpenAI Support
Copy code
model = openai.realtime.RealtimeModel.withAzure({
    baseURL: process.env.AZURE_OPENAI_ENDPOINT,
    azureDeployment: process.env.AZURE_OPENAI_DEPLOYMENT,
    apiKey: process.env.AZURE_OPENAI_API_KEY,
    entraToken: process.env.AZURE_OPENAI_ENTRA_TOKEN,
    instructions: "You are a helpful assistant.",
})
Complete Example
Copy code
from livekit.agents import JobContext, WorkerOptions, cli, multimodal
from livekit.plugins import openai

async def entrypoint(ctx: JobContext):
    await ctx.connect()
    model = openai.realtime.RealtimeModel(
        instructions="You are a helpful assistant.",
    )
    agent = multimodal.MultimodalAgent({ model })
    await agent.start(ctx.room)

if __name__ == "__main__":
    cli.run_app(WorkerOptions(entrypoint_fnc=entrypoint))
This model is seriously impressive - it lets you have super smooth speech-to-speech interactions without all the complicated translation steps. It's like having a conversation that just feels natural and keeps all those little emotional nuances. Got questions? I'm totally here to help you out! 🌟 Sources: RealtimeModel | LiveKit Agents | src/agent.ts | Realtime model integrations | LiveKit Docs | examples/src/multimodal_agent.ts