Spaces:
Runtime error
Runtime error
| """Run the LIVE ExecChat agent (current prompt + KB) through eval scenarios. | |
| Unlike agent-evals' snapshot-based runtime, this reuses the objects the running | |
| Streamlit app already built (db/reranker from the live Google Docs), so checks | |
| reflect the prompt/KB exactly as deployed right now. | |
| Tool-use scenarios are driven with mock tools that mirror the real tool names and | |
| record every call — no real API needed (works even when ENABLE_TOOLS=0). | |
| """ | |
| from __future__ import annotations | |
| from dataclasses import dataclass, field | |
| from typing import Any | |
| def make_mock_tools(call_log: list[dict]): | |
| """Return langchain tools with the agent's real names that record calls. | |
| ``call_log`` is mutated in place: each entry is {"name", "args"}. | |
| """ | |
| from langchain_core.tools import tool | |
| def _rec(name: str, args: dict, canned: str) -> str: | |
| call_log.append({"name": name, "args": dict(args)}) | |
| return canned | |
| def get_exec_quality_params(executive_id: str, period: str = "last 14 days") -> str: | |
| """Параметры качества/рейтинга исполнителя за период (по умолчанию 14 дней, макс 30).""" | |
| return _rec("get_exec_quality_params", {"executive_id": executive_id, "period": period}, | |
| "Качество: рейтинг 4.8/5, пунктуальность 96%, жалоб за период: 0.") | |
| def get_exec_orders_history(executive_id: str, period: str = "last 7 days") -> str: | |
| """История заказов исполнителя за период (по умолчанию 7 дней, макс 10).""" | |
| return _rec("get_exec_orders_history", {"executive_id": executive_id, "period": period}, | |
| "За период выполнено 6 заказов на 9 200 ₽, 1 заказ с опозданием.") | |
| def get_exec_next_orders(executive_id: str) -> str: | |
| """Ближайшие заказы исполнителя: услуги, адреса, оплата, комментарии клиента.""" | |
| return _rec("get_exec_next_orders", {"executive_id": executive_id}, | |
| "Завтра 10:00, поддерживающая уборка, ул. Ленина 5, оплата 1 800 ₽.") | |
| def get_exec_revenue_and_fines_feed(executive_id: str, period: str = "last 7 days") -> str: | |
| """Лента доходов и штрафов исполнителя за период (по умолчанию 7 дней, макс 10).""" | |
| return _rec("get_exec_revenue_and_fines_feed", {"executive_id": executive_id, "period": period}, | |
| "Доход за период: 9 200 ₽. Штрафов: 1 (опоздание, −300 ₽).") | |
| return [get_exec_next_orders, get_exec_quality_params, | |
| get_exec_revenue_and_fines_feed, get_exec_orders_history] | |
| class AgentRuntime: | |
| """Holds the live db/reranker/llms for the agent under test (exec_like).""" | |
| module: Any # the app_2 module (provides factories + constants) | |
| db: Any | |
| reranker: Any | |
| llm: Any # no-tools llm, reused for non-tool turns | |
| grounding_llm: Any | |
| query_rewriter: Any | |
| system_template: str # BASE prompt (TOOLS_INSTRUCTION appended per tool turn) | |
| app_supports_tools: bool = True # hard gate: does the agent under test have a tools system at all | |
| default_use_tools: bool = True # run-level default applied to scenarios set to "Авто" (use_tools=None) | |
| def run_turn(self, message: str, chat_history: list[tuple[str, str]] | None = None, | |
| use_tools: bool | None = None) -> dict: | |
| mod = self.module | |
| chat_history = list(chat_history or []) | |
| call_log: list[dict] = [] | |
| # ``use_tools`` is the per-scenario setting: True/False force it, None | |
| # ("Авто") falls back to the run-level default. A scenario that explicitly | |
| # asks for tools gets them regardless of the run-level default — the only | |
| # hard requirement is that the agent under test actually has a tools system. | |
| if use_tools is None: | |
| use_tools = self.default_use_tools | |
| tools_list = make_mock_tools(call_log) if (use_tools and self.app_supports_tools) else [] | |
| llm = mod.create_llm(0.0, tools_list) if tools_list else self.llm | |
| system_template = self.system_template | |
| if tools_list and hasattr(mod, "TOOLS_INSTRUCTION"): | |
| system_template = system_template + mod.TOOLS_INSTRUCTION | |
| text, rag, guard = mod.generate_response( | |
| self.db, self.reranker, self.query_rewriter, llm, self.grounding_llm, | |
| system_template, mod.USER_TEMPLATE, tools_list, message, chat_history, | |
| ) | |
| return {"text": text, "rag_chunks": rag, "guard_info": guard, "tool_calls": call_log} | |
| def run_dialog(self, user_turns: list[str], use_tools: bool | None = None) -> dict: | |
| history: list[tuple[str, str]] = [] | |
| turns = [] | |
| for msg in user_turns: | |
| res = self.run_turn(msg, history, use_tools=use_tools) | |
| turns.append({"user": msg, **res}) | |
| history.append(("user", msg)) | |
| history.append(("manager", res["text"])) | |
| return {"turns": turns, "final": turns[-1] if turns else None, | |
| "tool_calls": [c for t in turns for c in t["tool_calls"]]} | |
| def build_runtime_live(module, db, reranker, base_system_template: str, | |
| app_supports_tools: bool = True, | |
| default_use_tools: bool = True) -> AgentRuntime: | |
| """Assemble an AgentRuntime from the live app objects. | |
| ``module`` is the running app_2 module (provides create_llm / create_grounding_llm | |
| / create_query_rewriter / generate_response / constants). ``db`` and ``reranker`` | |
| are the cached objects from init_db()/init_reranker(); ``base_system_template`` is | |
| the prompt WITHOUT the tools instruction (run_turn appends it for tool turns). | |
| ``app_supports_tools`` is a hard capability gate (False for RAG-only agents like | |
| OnBoarding). ``default_use_tools`` is the run-level default for scenarios left on | |
| "Авто"; scenarios with an explicit use_tools=True get tools regardless of it. | |
| """ | |
| return AgentRuntime( | |
| module=module, | |
| db=db, | |
| reranker=reranker, | |
| llm=module.create_llm(0.0, []), | |
| grounding_llm=module.create_grounding_llm(), | |
| query_rewriter=module.create_query_rewriter(), | |
| system_template=(base_system_template or "").strip(), | |
| app_supports_tools=app_supports_tools, | |
| default_use_tools=default_use_tools, | |
| ) | |