agent-memory / agent.py
eriquesouza
Preparar deploy no Hugging Face Spaces.
2f56bf2
Raw
History Blame Contribute Delete
8.11 kB
import json
import os
import re
from json import JSONDecoder
from typing import List, Dict
from dotenv import load_dotenv
from pydantic import ValidationError
from langchain_core.messages import AIMessage, HumanMessage
from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder
from langchain_openai import ChatOpenAI
from models import AgentLLMOutput, ChatResponse, Memory
from memory_store import MemoryStore
load_dotenv()
def _openrouter_referer() -> str:
if host := os.getenv("SPACE_HOST"):
return host
if space_id := os.getenv("SPACE_ID"):
return f"https://huggingface.co/spaces/{space_id}"
return "http://localhost:8000"
SYSTEM_PROMPT = """\
You are a customer service assistant for an internet, TV, and telephony provider in Brazil.
This is a memory-observability PoC: you help the current customer resolve their situation using \
what past customers said in previous conversations — stored as memories below.
Memories are the customers' own words (or faithful paraphrase), NOT official policy, CRM data, \
or verified operational facts. Use them to empathize, recognize patterns, and suggest helpful \
approaches that worked before — with caution when memories conflict or may be outdated.
{memory_context}
─────────────────────────────────────────────
RESPONSE FORMAT — you MUST return valid JSON only, no other text:
{{
"response": "Your reply to the customer in Brazilian Portuguese (clear, empathetic, actionable)",
"memories_used": ["id1", "id2"],
"new_memories": [
{{
"content": "What the current customer said, preserved in their voice as closely as possible",
"type": "episodic|semantic|state|procedural (English only, not episodico/semantico)",
"context_tags": ["tag1", "tag2"],
"summary": "5-10 word summary for display"
}}
]
}}
─────────────────────────────────────────────
MEMORY TYPES (content is usually a past customer's statement):
- episodic → specific situation a customer reported ("my portability has been stuck for 5 days")
- semantic → recurring pattern from multiple customers ("new installs often question real speed")
- state → recent claim about current conditions ("the app won't load my boleto since yesterday")
- procedural → lesson from how support went ("asking cable vs Wi-Fi before sending a tech helped")
─────────────────────────────────────────────
GUIDELINES:
- Always respond to the customer in Brazilian Portuguese; be empathetic and avoid unexplained jargon
- Use only memories genuinely relevant to the current message
- List only the IDs of memories you actually drew on in your response
- When memories contradict each other, do NOT state uncertain things as fact; ask clarifying questions \
or acknowledge uncertainty
- Create new_memories only for noteworthy things the CURRENT customer said — keep their wording \
and tone; not every message needs a new memory
- Choose memory type based on what was said: one-off event (episodic), recurring theme (semantic), \
current-sounding situation (state), or insight about what helped/hurt in support (procedural)
- PoC limits: you have no access to billing, CRM, or network systems — guide with questions, \
logical troubleshooting steps, and reasonable next steps without inventing protocol numbers, \
discounts, stock levels, or coverage
- Respond ONLY with the JSON object — no preamble, no markdown fences
- Previous assistant turns in the chat history are plain-text summaries for context; \
your current reply must still be ONLY the JSON object, never duplicate the answer outside JSON
"""
OPENROUTER_API_URL = os.getenv("OPENROUTER_API_URL")
MODEL = os.getenv("OPENROUTER_MODEL")
PROMPT = ChatPromptTemplate.from_messages([
("system", SYSTEM_PROMPT),
MessagesPlaceholder("history"),
("human", "{input}"),
])
def _parse_agent_llm_output(text: str) -> AgentLLMOutput:
"""Accept strict JSON or model output with prose before/after the JSON object."""
text = (text or "").strip()
if not text:
raise ValueError("Empty LLM output")
decoder = JSONDecoder()
candidates: List[str] = [text]
for match in re.finditer(
r"```(?:json)?\s*(\{.*?\})\s*```", text, re.DOTALL | re.IGNORECASE
):
candidates.append(match.group(1))
for raw in candidates:
try:
return AgentLLMOutput.model_validate(json.loads(raw))
except (json.JSONDecodeError, ValueError, ValidationError):
continue
start = 0
while True:
brace = text.find("{", start)
if brace == -1:
break
try:
obj, _ = decoder.raw_decode(text, brace)
if isinstance(obj, dict) and "response" in obj:
return AgentLLMOutput.model_validate(obj)
except (json.JSONDecodeError, ValidationError):
pass
start = brace + 1
raise ValueError("No valid AgentLLMOutput JSON found in model response")
class Agent:
def __init__(self, memory_store: MemoryStore):
self.memory_store = memory_store
llm = ChatOpenAI(
base_url=OPENROUTER_API_URL,
api_key=os.getenv("OPENROUTER_API_KEY"),
model=MODEL,
max_tokens=1500,
timeout=60.0,
extra_body={
"chat_template_kwargs": {"enable_thinking": False},
"response_format": {"type": "json_object"},
},
default_headers={
"HTTP-Referer": _openrouter_referer(),
"X-Title": "Agent Memory Phase 1",
},
)
self.chain = PROMPT | llm
async def chat(
self,
message: str,
conversation_history: List[Dict[str, str]],
) -> ChatResponse:
relevant = self.memory_store.search(message, n_results=6)
state_mems = self.memory_store.search(message, n_results=3, type_filter="state")
seen: set = set()
candidates: List[Memory] = []
for m in relevant + state_mems:
if m.id not in seen:
seen.add(m.id)
candidates.append(m)
memory_context = self._format_memories(candidates)
history = [
HumanMessage(content=turn["content"])
if turn["role"] == "user"
else AIMessage(content=turn["content"])
for turn in conversation_history[-6:]
]
raw = await self.chain.ainvoke({
"memory_context": memory_context,
"input": message,
"history": history,
})
parsed = _parse_agent_llm_output(raw.content)
for mem_id in parsed.memories_used:
self.memory_store.update_access(mem_id)
new_memories_saved: List[Memory] = []
for nm in parsed.new_memories:
saved = self.memory_store.add_memory(
content=nm.content,
memory_type=nm.type.value,
source="agent",
context_tags=nm.context_tags,
summary=nm.summary,
)
new_memories_saved.append(saved)
return ChatResponse(
response=parsed.response,
memories_used=parsed.memories_used,
new_memories=new_memories_saved,
all_memories=self.memory_store.list_memories(),
)
def _format_memories(self, memories: List[Memory]) -> str:
if not memories:
return "(no memories available)"
lines = []
for m in memories:
tags = ", ".join(m.context_tags) if m.context_tags else "—"
lines.append(
f"[ID: {m.id}] [{m.type.upper()}] {m.content}\n"
f" tags: {tags} | score: {m.relevance_score:.2f} | accessed: {m.access_count}x"
)
return "\n\n".join(lines)