I am currently using `livekit-examples/agent-start...
# ask-ai
b
I am currently using
livekit-examples/agent-starter-react
template for the frontend, and Python agent for the backend with AWS Poly & Transcribe. When user speaks, the transcribed text in chat also tagged as if Agent speak in the chat windows. The participantInfo wrongly tagged to Agent name? How to fix this?
t
Thanks for reaching out, Gopi. I'm an AI assistant and I'm here to help. Give me a moment while I research this topic.
Can you share a bit more about how you’re setting or updating the participantInfo in your implementation? For instance, are you using any particular logic or API calls to assign participantInfo to messages? Understanding this will really help us get to the bottom of the issue together.
b
No, there is no custom implementation. I am using the same code base given in
livekit-examples/agent-starter-react
for the frontend. And the simple version python code base for the agent backend, below you refer the agent implementation in python:
Copy code
from livekit.plugins.turn_detector.english import EnglishModel
from dotenv import load_dotenv
import os
import json
import asyncio
from livekit import agents
from livekit.agents import AgentSession, Agent, RoomInputOptions, RoomOutputOptions
from livekit.plugins import (
    openai,
    aws,
    silero,
)
import logging
# from csv_tools import add_row, get_rows  # CSV tools as function_tools
from tools import send_mail

logging.basicConfig(
    level=logging.DEBUG,  # Change to INFO to reduce verbosity
    format="%(asctime)s [%(levelname)s] %(message)s",
    handlers=[logging.StreamHandler()]
)
logger = logging.getLogger(__name__)



load_dotenv()

llm_base_url = os.getenv("MGA_BASE_URL")
llm_api_key = os.getenv("MGA_BASE_KEY")
agent_name = os.getenv("AGENT_NAME")

# Load instructions from markdown file
def load_instructions(filename):
    instruction_file = os.path.join(os.path.dirname(__file__), filename)
    try:
        with open(instruction_file, 'r', encoding='utf-8') as f:
            return f.read()
    except FileNotFoundError:
        return "You are a helpful voice AI assistant."

class Assistant(Agent):
    def __init__(self, instructions: str) -> None:
        super().__init__(
            instructions=instructions,
            tools=[
                # add_row,
                # get_rows,
                send_mail
            ],
        )

async def entrypoint(ctx: agents.JobContext):
    DEFAULT_METADATA = {
        "participantName": "Bayer User",
        "participantIdentity": "<mailto:bayer.user@bayer.com|bayer.user@bayer.com>",
    }
    try:
        raw_meta_data = getattr(ctx.job, "metadata", "") or "{}"
        
        try:
            metadata = json.loads(raw_meta_data)
            <http://logger.info|logger.info>("Successfully parsed job metadata")

        except json.JSONDecodeError:
            metadata = DEFAULT_METADATA.copy()
            logger.warning("Invalid Job metadata, using default: %s", metadata)
    
        participantName = metadata.get("participantName", DEFAULT_METADATA["participantName"])
        participantIdentity = metadata.get("participantIdentity", DEFAULT_METADATA["participantIdentity"])

        <http://logger.info|logger.info>("Participant: name = %s; identity = %s", participantName, participantIdentity)

    except Exception as e:
        logger.exception("Unexpected error parsing ctx.job metadata")

    raw_agent_instructions = load_instructions("agent-instruction.md")
    formatted_agent_instructions = raw_agent_instructions.format(
        participant=participantName,
        participant_mail=participantIdentity)

    raw_session_instructions = load_instructions("session-instruction.md")
    formatted_session_instructions = raw_session_instructions.format(
        participant=participantName,
        participant_mail=participantIdentity)

    <http://logger.info|logger.info>("Room name: %s ", ctx.job.room.name)

    ctx.log_context_fields = {
        "room": ctx.job.room.name,
    }

    session = AgentSession(
        stt=aws.STT(language="en-US"),
        llm=openai.LLM(base_url=llm_base_url, api_key=llm_api_key, model="gpt-4o", temperature=0.4),
        tts=aws.TTS(voice="Ruth", speech_engine="generative", language="en-US", region="eu-central-1"),
        # min_speech_duration (seconregion=ds) : min duration of speech to be cut into chunks
        # min_silence_duration (seconds):  Silence that is required at the end of speech
        # max_buffered_speech (seconds) : Speech that is kept in its buffer
        # activation_threshold (lower to higher) : Sensitivity
        # Minimum detected speech duration before triggering an interruption.
        min_interruption_duration=1.0,
        vad=silero.VAD.load(min_speech_duration=0.07, min_silence_duration=0.6, max_buffered_speech=60, activation_threshold=0.7),
        turn_detection=EnglishModel(),
    )

    await session.start(
        room=ctx.room,
        agent=Assistant(formatted_agent_instructions),
        room_input_options=RoomInputOptions(),
        room_output_options=RoomOutputOptions(transcription_enabled=True)
    )

    await ctx.connect()

    await session.generate_reply(
        instructions=formatted_session_instructions
    )


if __name__ == "__main__":
    agents.cli.run_app(agents.WorkerOptions(
        entrypoint_fnc=entrypoint, agent_name=agent_name))
t
I don't have the answer you're looking for. You could also try asking your question: • in one of the other Slack channels or • to https://deepwiki.com/livekit/livekit_composite which is trained on all LiveKit source code If you find the answer, please post it here to help others!