// SPDX-FileCopyrightText: 2024 LiveKit, Inc. // S...
# ask-ai
w
// SPDX-FileCopyrightText: 2024 LiveKit, Inc. // SPDX-License-Identifier: Apache-2.0 import { type JobContext, type JobProcess, WorkerOptions, cli, defineAgent, llm, pipeline, } from '@livekit/agents'; import * as openai from '@livekit/agents-plugin-openai'; import * as silero from '@livekit/agents-plugin-silero'; import dotenv from 'dotenv'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { z } from 'zod'; const __dirname = path.dirname(fileURLToPath(import.meta.url)); const envPath = path.join(__dirname, '../.env.local'); dotenv.config({ path: envPath }); export default defineAgent({ prewarm: async (_proc_: JobProcess) => { console.log('Prewarming VAD...'); proc.userData.vad = await silero.VAD.load(); console.log('VAD prewarmed successfully'); }, entry: async (_ctx_: JobContext) => { console.log('=== AGENT ENTRY STARTED ==='); try { const vad = ctx.proc.userData.vad! as silero.VAD; console.log('VAD loaded from userData'); const initialContext = new llm.ChatContext().append({ role: llm.ChatRole.SYSTEM, text: 'You are Grace, a friendly AI assistant. Keep responses very short and conversational - aim for 1-2 sentences maximum. Be concise and natural.', }); console.log('Initial context created'); await ctx.connect(); console.log('Connected to context'); console.log('Waiting for participant...'); const participant = await ctx.waitForParticipant(); console.log(
Participant joined: ${participant.identity} (${participant.name})
); // Simple function context const fncCtx: llm.FunctionContext = { weather: { description: 'Get the weather in a location', parameters: z.object({ location: z.string().describe('The location to get the weather for'), }), execute: async ({ location }) => { console.log(
Getting weather for: ${_location_}
); try { const response = await fetch(
<https://wttr.in/${>_location_}?format=%C+%t
); if (!response.ok) { throw new Error(
Weather API returned status: ${response.status}
); } const weather = await response.text(); return `The weather in ${location} right now is ${weather}.`; } catch (error) { console.error('Weather API error:', error); return `Sorry, I couldn't get the weather for ${location} right now.`; } }, }, }; console.log('Function context created'); // Create agent with minimal configuration console.log('Creating VoicePipelineAgent...'); // Check API keys if (!process.env.OPENAI_API_KEY) { console.error('āŒ OPENAI_API_KEY is missing'); throw new Error('OPENAI_API_KEY is required'); } console.log('āœ… API keys validated'); const agent = new pipeline.VoicePipelineAgent( vad, new openai.STT({ model: 'whisper-1', apiKey: process.env.OPENAI_API_KEY, }), new openai.LLM({ model: 'gpt-4o-mini', apiKey: process.env.OPENAI_API_KEY, temperature: 0.7, }), new openai.TTS({ voice: 'alloy', apiKey: process.env.OPENAI_API_KEY, }), { chatCtx: initialContext, fncCtx: fncCtx, allowInterruptions: true, minEndpointingDelay: 600, // Reduced from 1500ms - faster turn detection interruptSpeechDuration: 200, // Reduced - faster interruption detection interruptMinWords: 0, } ); console.log('VoicePipelineAgent created'); // Set up event listeners let userSpeechStartTime: number | null = null; agent.on(pipeline.VPAEvent.AGENT_STARTED_SPEAKING, () => { console.log('šŸ—£ļø Agent started speaking'); }); agent.on(pipeline.VPAEvent.AGENT_STOPPED_SPEAKING, () => { console.log('šŸ”‡ Agent stopped speaking'); }); agent.on(pipeline.VPAEvent.USER_STARTED_SPEAKING, () => { userSpeechStartTime = Date.now(); console.log('šŸ‘¤ User started speaking'); }); agent.on(pipeline.VPAEvent.USER_STOPPED_SPEAKING, () => { const duration = userSpeechStartTime ? Date.now() - userSpeechStartTime : 0; console.log(
:bust_in_silhouette: User stopped speaking (duration: ${duration}ms)
); userSpeechStartTime = null; }); agent.on(pipeline.VPAEvent.USER_SPEECH_COMMITTED, (_msg_: llm.ChatMessage) => { console.log('šŸ“ User speech committed:', msg.content); }); agent.on(pipeline.VPAEvent.AGENT_SPEECH_COMMITTED, (_msg_: llm.ChatMessage) => { console.log('šŸ“ Agent speech committed:', msg.content); }); agent.on(pipeline.VPAEvent.AGENT_SPEECH_INTERRUPTED, () => { console.log('āš ļø Agent speech interrupted'); }); agent.on(pipeline.VPAEvent.FUNCTION_CALLS_COLLECTED, (calls) => { console.log('šŸ”§ Function calls collected:', calls); }); agent.on(pipeline.VPAEvent.FUNCTION_CALLS_FINISHED, (calls) => { console.log('āœ… Function calls finished:', calls); }); console.log('Event listeners set up'); // Start the agent console.log('Starting agent...'); agent.start(ctx.room, participant); console.log('āœ… Agent started successfully'); // Wait a moment for initialization console.log('Waiting for agent to fully initialize...'); await new Promise(resolve => setTimeout(resolve, 2000)); // Try initial greeting const greeting = `Hello ${participant.name || 'there'}! How can I help you today?`; console.log(
Attempting to say: "${greeting}"
); try { await agent.say(greeting, true); console.log('āœ… Initial greeting completed'); } catch (error) { console.error('āŒ Greeting failed:', error); console.log('Agent will continue without greeting'); } console.log('=== AGENT FULLY INITIALIZED ==='); console.log('Agent is now ready to handle user interactions...'); } catch (error) { console.error('āŒ Fatal error in agent entry:', error); throw error; } }, }); console.log('Starting CLI app...'); cli.runApp(new WorkerOptions({ agent: fileURLToPath(import.meta.url) })); i have my agent code here but the issue is whenever i use deepgram and elevenlabs in stead of using the alloy and whisper it jsut doesnt work no errors or anythin
t
Thanks for reaching out, Christian. I'm an AI assistant and I'm here to help. Give me a moment while I research this topic.
To integrate Deepgram and ElevenLabs with your LiveKit agent instead of OpenAI Whisper and Alloy, follow these steps: • *Install Required Plugins:*`import * as deepgram from '@livekit/agents-plugin-deepgram'; import * as elevenlabs from '@livekit/agents-plugin-elevenlabs';` • *Set Environment Variables:*`DEEPGRAM_API_KEY=<your deepgram api key> ELEVEN_API_KEY=<your elevenlabs api key>` Ensure these keys are correctly added to your environment configuration. • Update Your Agent Code: Replace the STT and TTS components in your
VoicePipelineAgent
constructor:
Copy code
const agent = new pipeline.VoicePipelineAgent(
  vad,
  new deepgram.STT(),        // Replace OpenAI Whisper with Deepgram STT
  new openai.LLM(),          // Keep your existing LLM
  new elevenlabs.TTS(),      // Replace Alloy with ElevenLabs TTS
  { chatCtx: initialContext, fncCtx },
);
• Import the Plugins: Ensure both plugins are imported at the top of your agent file:
Copy code
import * as deepgram from '@livekit/agents-plugin-deepgram';
import * as elevenlabs from '@livekit/agents-plugin-elevenlabs';
This setup should allow you to use Deepgram for speech recognition and ElevenLabs for text-to-speech synthesis. If the issue persists without errors, double-check the API keys and ensure the plugins are correctly installed and imported. Sources: Example Applications | livekit/livekit_composite | DeepWiki | Deepgram STT integration guide | LiveKit Docs | src/agent.ts | examples/src/outbound.ts | livekit-plugins/livekit-plugins-elevenlabs/README.md