cool-salesmen-60705
07/24/2025, 5:19 PMtall-airline-16756
07/24/2025, 5:19 PMtall-airline-16756
07/24/2025, 5:21 PMimport logging
from pathlib import Path
from dotenv import load_dotenv
from livekit.agents import JobContext, WorkerOptions, cli
from livekit.agents.voice import Agent, AgentSession
from livekit.plugins import openai, silero, deepgram, elevenlabs
load_dotenv()
class SimpleAgent(Agent):
def __init__(self) -> None:
super().__init__(
instructions="""
You are a translator. You translate the user's speech from English to French.
Every message you receive, translate it directly into French.
Do not respond with anything else but the translation.
""",
stt=deepgram.STT(),
llm=openai.LLM(model="gpt-4o"),
tts=elevenlabs.TTS(
model="eleven_multilingual_v2"
),
vad=silero.VAD.load()
)
async def on_enter(self):
self.session.generate_reply()
async def entrypoint(ctx: JobContext):
session = AgentSession()
await session.start(
agent=SimpleAgent(),
room=ctx.room
)
Advanced Multi-Language Captioning SystemFor a more sophisticated setup with live transcription and translation to multiple languages including French:
import asyncio
import logging
from livekit.agents import JobContext, WorkerOptions, cli, stt, llm, transcription
from livekit.plugins import openai, silero, deepgram
# Language configuration
languages = {
"en": {"code": "en", "name": "English", "flag": "🇺🇸"},
"fr": {"code": "fr", "name": "French", "flag": "🇫🇷"},
}
class Translator:
def __init__(self, room, lang):
self.room = room
self.lang = lang
self.context = llm.ChatContext().append(
role="system",
text=(
f"You are a translator for language: {lang.value}"
f"Your only response should be the exact translation of input text in the {lang.value} language."
),
)
self.llm = openai.LLM()
async def translate(self, message: str, track):
self.context.append(text=message, role="user")
stream = self.llm.chat(chat_ctx=self.context)
translated_message = ""
async for chunk in stream:
content = chunk.choices[0].delta.content
if content is None:
break
translated_message += content
# Publish transcription with translation
segment = rtc.TranscriptionSegment(
id=utils.misc.shortuuid("SG_"),
text=translated_message,
start_time=0,
end_time=0,
language=self.lang.name,
final=True,
)
transcription = rtc.Transcription(
self.room.local_participant.identity, track.sid, [segment]
)
await self.room.local_participant.publish_transcription(transcription)
async def entrypoint(job: JobContext):
stt_provider = deepgram.STT()
translators = {}
# Set up transcription forwarding and translation
async def _forward_transcription(stt_stream, stt_forwarder, track):
async for ev in stt_stream:
stt_forwarder.update(ev)
if ev.type == stt.SpeechEventType.FINAL_TRANSCRIPT:
message = ev.alternatives[0].text
# Translate to French for participants who requested it
for translator in translators.values():
asyncio.create_task(translator.translate(message, track))
Required DependenciesInstall the necessary packages:
livekit-agents
livekit-plugins-openai
livekit-plugins-deepgram
livekit-plugins-silero
python-dotenv
Environment SetupYou’ll need API keys for both services:
• OpenAI API Key from the OpenAI platform
• Deepgram API Key from the Deepgram console
Set these in your .env file:
OPENAI_API_KEY=your_openai_key
DEEPGRAM_API_KEY=your_deepgram_key
Key FeaturesThis setup provides:
1. Real-time transcription using Deepgram STT
2. Translation to French using OpenAI GPT-4o
3. Live captions published to participants
4. Multi-language support with French explicitly supported
5. Voice-to-voice translation when combined with TTS
This approach simplifies the process while ensuring accurate transcription and translation in French.
Sources: pipeline-llm/transcription_node.py | server/requirements.txt | Vision Agent Quickstart | LiveKit Docs