diff --git a/apps/docs/integrations/livekit.mdx b/apps/docs/integrations/livekit.mdx index 8a85a4d3..5b090084 100644 --- a/apps/docs/integrations/livekit.mdx +++ b/apps/docs/integrations/livekit.mdx @@ -93,6 +93,8 @@ if __name__ == "__main__": `preload` puts the caller's profile into the first turn, so the greeting can use it. `attach` stores the conversation after each agent reply, and stores anything left when the session closes, including a caller who hangs up mid-turn. `SupermemoryAgent` recalls inside `llm_node` before each reply and adds the memory tools. That covers voice turns, text turns from `generate_reply` or `session.run`, and LiveKit's preemptive generation. Each user message is recalled once, and tool follow-ups reuse it. +A runnable version, with a greeting that uses the caller's memory, is in the [example agent](https://github.com/supermemoryai/supermemory/tree/main/packages/livekit-sdk-python/examples). + `agent_name` turns on explicit dispatch, which is how job metadata reaches the agent. Remove it to join every new room automatically, for example when testing in the [Agents Playground](https://agents-playground.livekit.io). ## Your own agent diff --git a/packages/livekit-sdk-python/README.md b/packages/livekit-sdk-python/README.md index b88bd531..c6d918e1 100644 --- a/packages/livekit-sdk-python/README.md +++ b/packages/livekit-sdk-python/README.md @@ -71,6 +71,8 @@ if __name__ == "__main__": agents.cli.run_app(server) ``` +A runnable version with a greeting that uses memory is in [examples/](examples/). + If you already have an `Agent` subclass, pass `tools=memory.tools()` and recall from `llm_node`: ```python diff --git a/packages/livekit-sdk-python/examples/.env.example b/packages/livekit-sdk-python/examples/.env.example new file mode 100644 index 00000000..3bda6f73 --- /dev/null +++ b/packages/livekit-sdk-python/examples/.env.example @@ -0,0 +1,14 @@ +# LiveKit Cloud project: Settings > API keys. The same keys cover STT, LLM, and TTS +# through LiveKit Inference. +LIVEKIT_URL=wss://your-project.livekit.cloud +LIVEKIT_API_KEY= +LIVEKIT_API_SECRET= + +# https://console.supermemory.ai +SUPERMEMORY_API_KEY= + +# Optional: pin every call to one caller while testing. +# SUPERMEMORY_CONTAINER_TAG=demo_caller + +# Optional: require explicit dispatch, so your backend can pass {"container_tag": ...}. +# LIVEKIT_AGENT_NAME=memory-agent diff --git a/packages/livekit-sdk-python/examples/README.md b/packages/livekit-sdk-python/examples/README.md new file mode 100644 index 00000000..85cfea14 --- /dev/null +++ b/packages/livekit-sdk-python/examples/README.md @@ -0,0 +1,38 @@ +# Voice agent example + +A LiveKit voice agent that remembers each caller. Tell it something on one call, hang up, and it knows it on the next. + +## Setup + +```bash +pip install supermemory-livekit python-dotenv +cp .env.example .env +``` + +Fill in `.env` with your LiveKit Cloud project keys and a [Supermemory API key](https://console.supermemory.ai). LiveKit Inference provides speech-to-text, the LLM, and text-to-speech, so no other keys are needed. + +From a checkout of this repo, install the local package instead: `pip install -e .. python-dotenv`. + +## Talk to it + +In your terminal, through your mic: + +```bash +python voice_agent.py console +``` + +In the browser: run `python voice_agent.py dev`, open the [Agents Playground](https://agents-playground.livekit.io), and connect to your project. + +## Try memory + +1. Say "My name is Priya, and please remember I'm vegetarian." +2. Hang up and start a new call. +3. The agent greets you by name. Ask "What should I order for dinner?" + +Memory is scoped per caller: + +- Console mode uses the container tag `console_user`. +- In rooms, the agent uses the participant attribute `supermemory_container_tag`, or else the participant identity. The Playground gives each session a new identity, so set `SUPERMEMORY_CONTAINER_TAG` in `.env` to keep one caller across Playground calls. +- In production, dispatch the agent from your backend with `{"container_tag": ""}` as job metadata, and set `LIVEKIT_AGENT_NAME`. + +Each call is stored as one document, and facts the caller asks it to remember are saved right away. See the [integration docs](https://supermemory.ai/docs/integrations/livekit) for configuration. diff --git a/packages/livekit-sdk-python/examples/voice_agent.py b/packages/livekit-sdk-python/examples/voice_agent.py new file mode 100644 index 00000000..f6d4a7cf --- /dev/null +++ b/packages/livekit-sdk-python/examples/voice_agent.py @@ -0,0 +1,71 @@ +"""Voice agent that remembers each caller across calls. + +Run from this folder after filling in .env (see README.md): + + python voice_agent.py console # talk through your mic in the terminal + python voice_agent.py dev # join LiveKit rooms, e.g. from the Agents Playground +""" + +import json +import logging +import os + +from dotenv import load_dotenv +from livekit import agents +from livekit.agents import AgentServer, AgentSession, ChatContext, JobContext +from supermemory_livekit import SupermemoryAgent, SupermemoryLiveKit + +load_dotenv() +logger = logging.getLogger("voice-agent") + +INSTRUCTIONS = ( + "You are a friendly voice assistant. You remember this caller across calls. " + "Use what you know naturally, and do not mention the memory system. " + "When the caller asks you to remember something, call remember. Keep replies short." +) + +server = AgentServer() + + +# An agent_name turns on explicit dispatch, which is how job metadata reaches the agent. +# Leave it empty to join every new room, which the Agents Playground needs. +@server.rtc_session(agent_name=os.getenv("LIVEKIT_AGENT_NAME", "")) +async def entrypoint(ctx: JobContext): + memory = SupermemoryLiveKit(api_key=os.environ["SUPERMEMORY_API_KEY"]) + + # Your backend can pass the caller id in dispatch metadata. SUPERMEMORY_CONTAINER_TAG + # pins every call to one caller while testing. Console mode has no remote participant. + metadata = json.loads(ctx.job.metadata or "{}") + container_tag = metadata.get("container_tag") or os.getenv("SUPERMEMORY_CONTAINER_TAG") + if not container_tag and ctx.is_fake_job(): + container_tag = "console_user" + + await ctx.connect() + if container_tag: + memory.bind(container_tag=container_tag, session_id=ctx.room.name) + else: + participant = await ctx.wait_for_participant() + memory.bind(participant=participant, session_id=ctx.room.name) + logger.info("memory scoped to container tag %s", memory.container_tag) + + # The caller's profile goes into the first turn, so the greeting can use it. + chat_ctx = ChatContext() + await memory.preload(chat_ctx) + + session = AgentSession( + stt="deepgram/nova-3:en", + llm="openai/gpt-4.1-mini", + tts="cartesia/sonic-3", + ) + memory.attach(session) + await session.start( + room=ctx.room, + agent=SupermemoryAgent(memory, chat_ctx=chat_ctx, instructions=INSTRUCTIONS), + ) + await session.generate_reply( + instructions="Greet the caller. If you already know them, welcome them back by name." + ) + + +if __name__ == "__main__": + agents.cli.run_app(server)