calm-article-62769
09/24/2025, 6:50 PMtall-airline-16756
09/24/2025, 6:50 PMtall-airline-16756
09/24/2025, 6:51 PMllm_node or transcription_node. You can access the full conversation history anytime through the session.history property. Plus, you can save these transcripts to files using add_shutdown_callback, or get instant updates via events like conversation_item_added and user_input_transcribed.
Just so you know, recording starts as soon as an agent joins the room and automatically captures all audio and video shared there. It will stop once everyone has left the session. If you have any questions or need further help with this, I’m here for you!
Sources: Session recording and transcripts | LiveKit Docs | Recording agent sessions | LiveKit Docscalm-article-62769
09/24/2025, 6:53 PMtall-airline-16756
09/24/2025, 6:54 PMcalm-article-62769
09/24/2025, 6:55 PMtall-airline-16756
09/24/2025, 6:56 PMcalm-article-62769
09/24/2025, 6:59 PMimport asyncio
import logging
from dotenv import load_dotenv
from livekit import rtc
from livekit.agents import JobContext, WorkerOptions, cli, get_job_context
from livekit.agents.llm import ChatContext, ChatMessage, ImageContent
from livekit.agents.voice import Agent, AgentSession
from livekit.plugins import google, openai, silero, deepgram, elevenlabs
logger = logging.getLogger("vision-agent")
logger.setLevel(<http://logging.INFO|logging.INFO>)
load_dotenv(".env.local")
class VisionAgent(Agent):
def __init__(self) -> None:
self._latest_frame = None
self._video_stream = None
self._tasks = []
super().__init__(
instructions="""
You are an assistant communicating through voice with vision capabilities.
You can see what the user is showing you through their camera.
Don't use any unpronouncable characters.
""",
# Try with realtime models of google. But not able see the screen with this.
# llm=google.beta.realtime.RealtimeModel(model = 'gemini-2.0-flash-live-001'),
# Using gemini flash llm and then STT and TTS models from other Elevnlabs and Deepgram. Screenshare working fine here.
stt=deepgram.STT(model="nova-2"),
llm=google.LLM(model="gemini-2.5-flash", temperature=0.25),
tts=elevenlabs.TTS(),
vad=silero.VAD.load(),
)
async def on_enter(self):
room = get_job_context().room
# Find the first video track (if any) from the remote participant
if room.remote_participants:
remote_participant = list(room.remote_participants.values())[0]
video_tracks = [
publication.track
for publication in list(remote_participant.track_publications.values())
if publication.track
and publication.track.kind == rtc.TrackKind.KIND_VIDEO
]
if video_tracks:
self._create_video_stream(video_tracks[0])
# Watch for new video tracks not yet published
@room.on("track_subscribed")
def on_track_subscribed(
track: rtc.Track,
publication: rtc.RemoteTrackPublication,
participant: rtc.RemoteParticipant,
):
if track.kind == rtc.TrackKind.KIND_VIDEO:
self._create_video_stream(track)
async def on_user_turn_completed(
self, turn_ctx: ChatContext, new_message: ChatMessage
) -> None:
# Add the latest video frame, if any, to the new message
if self._latest_frame:
new_message.content.append(ImageContent(image=self._latest_frame))
self._latest_frame = None
# Helper method to buffer the latest video frame from the user's track
def _create_video_stream(self, track: rtc.Track):
# Close any existing stream (we only want one at a time)
if self._video_stream is not None:
self._video_stream.close()
# Create a new stream to receive frames
self._video_stream = rtc.VideoStream(track)
async def read_stream():
async for event in self._video_stream:
# Store the latest frame for use later
self._latest_frame = event.frame
# Store the async task
task = asyncio.create_task(read_stream())
task.add_done_callback(lambda t: self._tasks.remove(t))
self._tasks.append(task)
async def entrypoint(ctx: JobContext):
session = AgentSession()
await session.start(agent=VisionAgent(), room=ctx.room)
if __name__ == "__main__":
cli.run_app(WorkerOptions(entrypoint_fnc=entrypoint))tall-airline-16756
09/24/2025, 6:59 PM