Spaces:
Sleeping
Sleeping
| import logging | |
| from dotenv import load_dotenv | |
| from livekit import agents | |
| from livekit.agents import Agent, AgentServer, AgentSession, JobContext, room_io | |
| # from livekit.plugins import noise_cancellation, silero | |
| from livekit.plugins import noise_cancellation, silero | |
| from livekit.agents import llm, stt, tts, inference | |
| from livekit.plugins.turn_detector.multilingual import MultilingualModel | |
| from livekit.agents import AgentStateChangedEvent, MetricsCollectedEvent, metrics | |
| import time | |
| import httpx | |
| from livekit.agents import function_tool, RunContext, ToolError | |
| from livekit.agents import mcp | |
| from livekit.plugins import cartesia | |
| load_dotenv() | |
| logger = logging.getLogger(__name__) | |
| # Define your agent's behavior by extending the Agent class | |
| class Assistant(Agent): | |
| def __init__(self) -> None: | |
| super().__init__( | |
| instructions=( | |
| "Your default and primary language is Indonesian. Always respond in Indonesian unless prompted otherwise. " | |
| "If the user addresses you in English, you must immediately switch to English and respond in English." | |
| "Maintain the language chosen by the user throughout the conversation until they switch back to the other language." | |
| "Be professional, friendly, and accurate in both languages." | |
| "If the user asks a question in Indonesian, provide a comprehensive, clear, and natural-sounding Indonesian response." | |
| "If the user asks a question in English, provide a comprehensive, clear, and natural-sounding English response." | |
| "Always maintain the persona regardless of the language being used." | |
| "Help the caller fix issues without rambling, and keep replies under 3 sentences. " | |
| "You can also look up the weather if asked." | |
| "LiveKit by searching the documentation. When users ask about LiveKit " | |
| "features, APIs, or how to build something, use the docs search tools " | |
| "to find accurate information." | |
| ), | |
| ) | |
| # The @function_tool decorator registers this method as a tool the LLM can call | |
| async def lookup_weather( | |
| self, | |
| context: RunContext, # Gives access to the session, speech handle, and user data | |
| location: str, # Type hints help the LLM understand what arguments to pass | |
| ) -> dict: | |
| """Look up current weather for a location. | |
| Args: | |
| location: City name or location to get weather for. | |
| """ | |
| # The docstring above becomes the tool description the LLM sees | |
| # when deciding which tool to call | |
| async with httpx.AsyncClient() as client: | |
| # First, geocode the location to get coordinates | |
| geo_response = await client.get( | |
| "https://geocoding-api.open-meteo.com/v1/search", | |
| params={"name": location, "count": 1} | |
| ) | |
| geo_data = geo_response.json() | |
| if not geo_data.get("results"): | |
| raise ToolError(f"Could not find location: {location}") | |
| lat = geo_data["results"][0]["latitude"] | |
| lon = geo_data["results"][0]["longitude"] | |
| place_name = geo_data["results"][0]["name"] | |
| # Get current weather for those coordinates | |
| weather_response = await client.get( | |
| "https://api.open-meteo.com/v1/forecast", | |
| params={ | |
| "latitude": lat, | |
| "longitude": lon, | |
| "current": "temperature_2m,weather_code", | |
| "temperature_unit": "fahrenheit" | |
| } | |
| ) | |
| weather = weather_response.json() | |
| # Return a dict with the weather data | |
| # The LLM will use this to form a natural response | |
| return { | |
| "location": place_name, | |
| "temperature_f": weather["current"]["temperature_2m"], | |
| "conditions": weather["current"]["weather_code"] | |
| } | |
| server = AgentServer() | |
| # The entrypoint function runs when a participant joins the room | |
| async def entrypoint(ctx: JobContext): | |
| # Configure the voice pipeline with STT, LLM, TTS, and VAD providers | |
| session = AgentSession( | |
| # LLM with fallback: OpenAI primary, Gemini backup | |
| llm=llm.FallbackAdapter( | |
| [ | |
| inference.LLM(model="openai/gpt-4.1-mini"), | |
| inference.LLM(model="google/gemini-2.5-flash"), | |
| ] | |
| ), | |
| # STT with fallback: AssemblyAI primary, Deepgram backup | |
| stt=stt.FallbackAdapter( | |
| [ | |
| inference.STT.from_model_string("assemblyai/universal-streaming:en"), | |
| inference.STT.from_model_string("deepgram/nova-3"), | |
| ] | |
| ), | |
| # TTS with fallback: Cartesia primary, Inworld backup | |
| tts=tts.FallbackAdapter( | |
| [ | |
| cartesia.TTS( | |
| model="sonic-3", | |
| voice="f786b574-daa5-4673-aa0c-cbe3e8534c02", | |
| ), | |
| #inference.TTS.from_model_string("cartesia/sonic-3:72a1dea2-302f-4f6b-8936-e28f8d1a51ba"), | |
| inference.TTS.from_model_string("inworld/inworld-tts-1"), | |
| ] | |
| ), | |
| vad=silero.VAD.load(), | |
| turn_detection=MultilingualModel(), | |
| preemptive_generation=True, | |
| mcp_servers=[ | |
| mcp.MCPServerHTTP(url="https://docs.livekit.io/mcp") | |
| ] | |
| ) | |
| # Aggregate data across all conversation turns | |
| usage_collector = metrics.UsageCollector() | |
| # Track End of Utterance timing (when turn detector decides user finished speaking) | |
| last_eou_metrics: metrics.EOUMetrics | None = None | |
| def _on_metrics_collected(ev: MetricsCollectedEvent): | |
| nonlocal last_eou_metrics | |
| # Capture EOU metrics for TTFA calculation | |
| if ev.metrics.type == "eou_metrics": | |
| last_eou_metrics = ev.metrics | |
| # Log each metric as it arrives and add to usage collector | |
| metrics.log_metrics(ev.metrics) | |
| usage_collector.collect(ev.metrics) | |
| async def log_usage(): | |
| # Print per-session summary (tokens, audio duration, costs) | |
| summary = usage_collector.get_summary() | |
| logger.info("Usage summary: %s", summary) | |
| ctx.add_shutdown_callback(log_usage) | |
| def _on_agent_state_changed(ev: AgentStateChangedEvent): | |
| if ev.new_state == "speaking": | |
| if last_eou_metrics: | |
| # Calculate time since user finished speaking | |
| elapsed = time.time() - last_eou_metrics.timestamp | |
| logger.info(f"Time to first audio: {elapsed:.3f}s") | |
| # Start the session with noise cancellation enabled | |
| await session.start( | |
| agent=Assistant(), | |
| room=ctx.room, | |
| room_options=room_io.RoomOptions( | |
| audio_input=room_io.AudioInputOptions( | |
| noise_cancellation=noise_cancellation.BVC(), # Background voice cancellation | |
| ), | |
| ), | |
| record=False, # Enable recording for metrics and debugging | |
| ) | |
| if __name__ == "__main__": | |
| logging.basicConfig(level=logging.INFO) | |
| agents.cli.run_app(server) |