| import json
|
| import os
|
| import traceback
|
| from pathlib import Path
|
| from typing import Any, Dict, List, Optional, Union
|
|
|
| import gradio as gr
|
|
|
|
|
|
|
| from PackedLLM import PackedLLMRunner
|
|
|
|
|
| CHECKPOINT_PATH = os.getenv("PACKEDLLM_CHECKPOINT", "PackedLLM.pt")
|
| MAP_LOCATION = os.getenv("PACKEDLLM_MAP_LOCATION", "cpu")
|
| BOT_ID = os.getenv("PACKEDLLM_BOT_ID", "pip")
|
| USER_ID = os.getenv("PACKEDLLM_USER_ID", "space_user")
|
| DEFAULT_VENV_ID = os.getenv("PACKEDLLM_VENV_ID", "space_default")
|
|
|
|
|
| _runner: Optional[PackedLLMRunner] = None
|
|
|
|
|
| def _format_any(value: Any) -> str:
|
| if isinstance(value, str):
|
| return value
|
| try:
|
| return json.dumps(value, indent=2, ensure_ascii=False, default=str)
|
| except Exception:
|
| return repr(value)
|
|
|
|
|
| def _parse_json(text: str, fallback: Any = None) -> Any:
|
| text = (text or "").strip()
|
| if not text:
|
| return fallback
|
| return json.loads(text)
|
|
|
|
|
| def _parse_tags(text: str) -> List[str]:
|
| raw = (text or "").strip()
|
| if not raw:
|
| return []
|
| return [t.strip() for t in raw.split(",") if t.strip()]
|
|
|
|
|
| def get_runner() -> PackedLLMRunner:
|
| global _runner
|
| if _runner is None:
|
| _runner = PackedLLMRunner(
|
| checkpoint_path=CHECKPOINT_PATH,
|
| map_location=MAP_LOCATION,
|
| bot_id=BOT_ID,
|
| user_id=USER_ID,
|
| warmup=False,
|
| verbose=True,
|
| )
|
| return _runner
|
|
|
|
|
| def chat_fn(
|
| message: str,
|
| history: List[Dict[str, str]],
|
| image: Optional[str],
|
| deep_think: bool,
|
| fast_think: bool,
|
| stream: bool,
|
| ):
|
| history = history or []
|
| message = (message or "").strip()
|
|
|
| if not message and not image:
|
| yield history, history, ""
|
| return
|
|
|
| user_content = message
|
| if image:
|
| if user_content:
|
| user_content += f"\n\n[Image attached: {image}]"
|
| else:
|
| user_content = f"[Image attached: {image}]"
|
|
|
| history = history + [{"role": "user", "content": user_content}]
|
| yield history, history, ""
|
|
|
| try:
|
| runner = get_runner()
|
| result = runner.chat(
|
| message,
|
| image=image,
|
| stream=stream,
|
| deep_think=deep_think,
|
| fast_think=fast_think,
|
| )
|
|
|
| if stream and hasattr(result, "__iter__") and not isinstance(result, (str, bytes, dict, list, tuple)):
|
| assembled = ""
|
| for chunk in result:
|
| if isinstance(chunk, dict) and "content" in chunk:
|
| chunk_text = str(chunk["content"])
|
| else:
|
| chunk_text = str(chunk)
|
| assembled += chunk_text
|
| live_history = history + [{"role": "assistant", "content": assembled}]
|
| yield live_history, live_history, ""
|
|
|
| history = history + [{"role": "assistant", "content": assembled or ""}]
|
| yield history, history, ""
|
| else:
|
| history = history + [{"role": "assistant", "content": _format_any(result)}]
|
| yield history, history, ""
|
|
|
| except Exception as exc:
|
| err = traceback.format_exc()
|
| history = history + [
|
| {
|
| "role": "assistant",
|
| "content": f"**Error:** {exc}\n\n```text\n{err}\n```",
|
| }
|
| ]
|
| yield history, history, ""
|
|
|
|
|
| def run_expert(
|
| expert_name: str,
|
| prompt: str,
|
| image: Optional[str],
|
| character_card: str,
|
| logic_mode: str,
|
| tools_json: str,
|
| ) -> str:
|
| runner = get_runner()
|
| prompt = (prompt or "").strip()
|
|
|
| if not prompt and expert_name not in {"vision", "tool"}:
|
| raise gr.Error("Please enter a prompt.")
|
|
|
| kwargs: Dict[str, Any] = {}
|
|
|
| if expert_name == "role" and character_card.strip():
|
| kwargs["character_card"] = character_card.strip()
|
|
|
| if expert_name == "logic":
|
| kwargs["mode"] = logic_mode or "deep_then_answer"
|
|
|
| if expert_name == "vision":
|
| if not image:
|
| raise gr.Error("Please upload an image for the Vision expert.")
|
| kwargs["image"] = image
|
|
|
| if expert_name == "tool":
|
| tools = _parse_json(tools_json, fallback=None)
|
| if not tools:
|
| tools = [
|
| {
|
| "name": "noop",
|
| "description": "No-op demo tool.",
|
| "parameters": {"type": "object", "properties": {}},
|
| }
|
| ]
|
| return _format_any(runner.tool(prompt, tools=tools))
|
|
|
| if expert_name == "head":
|
| return _format_any(runner.head(prompt, image=image, **kwargs))
|
| if expert_name == "creative":
|
| return _format_any(runner.creative(prompt, **kwargs))
|
| if expert_name == "code":
|
| return _format_any(runner.code(prompt, **kwargs))
|
| if expert_name == "logic":
|
| return _format_any(runner.logic(prompt, **kwargs))
|
| if expert_name == "math":
|
| return _format_any(runner.math(prompt, **kwargs))
|
| if expert_name == "translate":
|
| return _format_any(runner.translate(prompt, **kwargs))
|
| if expert_name == "affect":
|
| return _format_any(runner.affect(prompt, **kwargs))
|
| if expert_name == "role":
|
| return _format_any(runner.role(prompt, **kwargs))
|
| if expert_name == "vision":
|
| return _format_any(runner.vision(prompt, image=image, **kwargs))
|
| if expert_name == "web":
|
| return _format_any(runner.web(prompt, **kwargs))
|
| if expert_name == "action":
|
| return _format_any(runner.action(prompt, **kwargs))
|
|
|
| raise gr.Error(f"Unknown expert: {expert_name}")
|
|
|
|
|
| def store_memory(text: str, tags: str, importance: float) -> str:
|
| runner = get_runner()
|
| result = runner.memory_store(
|
| text=text.strip(),
|
| tags=_parse_tags(tags),
|
| importance=float(importance),
|
| )
|
| return _format_any(result)
|
|
|
|
|
| def recall_memory(query: str, top_k: int) -> str:
|
| runner = get_runner()
|
| result = runner.memory_recall(query.strip(), top_k=int(top_k))
|
| return _format_any(result)
|
|
|
|
|
| def refresh_profiles() -> tuple[str, str]:
|
| runner = get_runner()
|
| return (
|
| json.dumps(runner.get_user_profile(), indent=2, ensure_ascii=False, default=str),
|
| json.dumps(runner.get_bot_profile(), indent=2, ensure_ascii=False, default=str),
|
| )
|
|
|
|
|
| def apply_profiles(user_profile_json: str, bot_profile_json: str) -> str:
|
| runner = get_runner()
|
| user_updates = _parse_json(user_profile_json, fallback={})
|
| bot_updates = _parse_json(bot_profile_json, fallback={})
|
|
|
| if not isinstance(user_updates, dict):
|
| raise gr.Error("User profile JSON must be an object.")
|
| if not isinstance(bot_updates, dict):
|
| raise gr.Error("Bot profile JSON must be an object.")
|
|
|
| if user_updates:
|
| runner.set_user_profile(user_updates)
|
| if bot_updates:
|
| runner.set_bot_profile(bot_updates)
|
|
|
| return "Profiles updated."
|
|
|
|
|
| def warmup_runner(include_web: bool, include_vision: bool, include_action: bool) -> str:
|
| runner = get_runner()
|
| report = runner.warmup(
|
| include_web=include_web,
|
| include_vision=include_vision,
|
| include_action=include_action,
|
| )
|
| return _format_any(report)
|
|
|
|
|
| def get_status() -> str:
|
| runner = get_runner()
|
| return _format_any(runner.status())
|
|
|
|
|
| def reload_expert(expert_name: str) -> str:
|
| runner = get_runner()
|
| result = runner.reload_expert(expert_name)
|
| return _format_any(result)
|
|
|
|
|
| def unload_expert(expert_name: str) -> str:
|
| runner = get_runner()
|
| runner.unload_expert(expert_name)
|
| return f"Unloaded {expert_name}."
|
|
|
|
|
| def unload_all() -> str:
|
| runner = get_runner()
|
| runner.unload_all()
|
| return "Unloaded all experts."
|
|
|
|
|
| def save_checkpoint(path: str) -> str:
|
| runner = get_runner()
|
| path = (path or "").strip()
|
| if path:
|
| runner.save(path)
|
| return f"Saved checkpoint to {path}"
|
| runner.save()
|
| return f"Saved checkpoint to {CHECKPOINT_PATH}"
|
|
|
|
|
| def run_code(
|
| code: str,
|
| venv_id: str,
|
| requirements_text: str,
|
| timeout: int,
|
| max_ram_mb: int,
|
| ) -> str:
|
| runner = get_runner()
|
| reqs = [line.strip() for line in (requirements_text or "").splitlines() if line.strip()]
|
| result = runner.run_code(
|
| code=code,
|
| venv_id=venv_id or DEFAULT_VENV_ID,
|
| requirements=reqs or None,
|
| timeout=int(timeout),
|
| max_ram_mb=int(max_ram_mb),
|
| ensure_venv=True,
|
| )
|
| return _format_any(result)
|
|
|
|
|
| def web_search(query: str, deep_search: bool) -> str:
|
| runner = get_runner()
|
| result = runner.web_search(query.strip(), deep_search=deep_search)
|
| return _format_any(result)
|
|
|
|
|
| with gr.Blocks(title="PackedLLM Demo", theme=gr.themes.Soft()) as demo:
|
| gr.Markdown(
|
| f"""
|
| # PackedLLM
|
|
|
| Loaded checkpoint: `{CHECKPOINT_PATH}`
|
|
|
| This demo exposes the main chat pipeline, direct expert calls, memory, web search, code execution, and system controls.
|
| """
|
| )
|
|
|
| with gr.Tabs():
|
| with gr.Tab("Chat"):
|
| chatbot = gr.Chatbot(type="messages", height=600, label="PackedLLM Chat")
|
| history_state = gr.State([])
|
|
|
| with gr.Row():
|
| image_in = gr.Image(
|
| type="filepath",
|
| label="Optional image for vision-enabled turns",
|
| )
|
|
|
| prompt_in = gr.Textbox(
|
| label="Message",
|
| placeholder="Ask PackedLLM anything...",
|
| lines=3,
|
| )
|
|
|
| with gr.Row():
|
| deep_think_in = gr.Checkbox(value=False, label="Deep think")
|
| fast_think_in = gr.Checkbox(value=False, label="Fast think")
|
| stream_in = gr.Checkbox(value=True, label="Stream")
|
|
|
| with gr.Row():
|
| send_btn = gr.Button("Send", variant="primary")
|
| clear_btn = gr.Button("Clear")
|
|
|
| send_btn.click(
|
| chat_fn,
|
| inputs=[prompt_in, history_state, image_in, deep_think_in, fast_think_in, stream_in],
|
| outputs=[chatbot, history_state, prompt_in],
|
| )
|
| prompt_in.submit(
|
| chat_fn,
|
| inputs=[prompt_in, history_state, image_in, deep_think_in, fast_think_in, stream_in],
|
| outputs=[chatbot, history_state, prompt_in],
|
| )
|
|
|
| def clear_chat():
|
| return [], [], ""
|
|
|
| clear_btn.click(clear_chat, outputs=[chatbot, history_state, prompt_in])
|
|
|
| with gr.Tab("Experts"):
|
| gr.Markdown("Call individual experts directly.")
|
|
|
| with gr.Row():
|
| expert = gr.Dropdown(
|
| choices=[
|
| "head",
|
| "creative",
|
| "code",
|
| "logic",
|
| "math",
|
| "translate",
|
| "affect",
|
| "role",
|
| "vision",
|
| "tool",
|
| "web",
|
| "action",
|
| ],
|
| value="head",
|
| label="Expert",
|
| )
|
| logic_mode = gr.Dropdown(
|
| choices=["deep_then_answer", "answer_only", "deep_only"],
|
| value="deep_then_answer",
|
| label="Logic mode",
|
| )
|
|
|
| expert_prompt = gr.Textbox(label="Prompt", lines=6, placeholder="Enter a prompt for the chosen expert.")
|
| expert_image = gr.Image(type="filepath", label="Image for Vision expert (optional)")
|
| character_card = gr.Textbox(
|
| label="Character card for Role expert",
|
| lines=4,
|
| placeholder="You are Pip, a direct and slightly sarcastic assistant.",
|
| )
|
| tools_json = gr.Textbox(
|
| label="Tools JSON for Tool expert",
|
| lines=8,
|
| placeholder='[{"name":"noop","description":"No-op demo tool.","parameters":{"type":"object","properties":{}}}]',
|
| )
|
| expert_out = gr.Textbox(label="Output", lines=18)
|
|
|
| run_expert_btn = gr.Button("Run expert", variant="primary")
|
| run_expert_btn.click(
|
| run_expert,
|
| inputs=[expert, expert_prompt, expert_image, character_card, logic_mode, tools_json],
|
| outputs=[expert_out],
|
| )
|
|
|
| with gr.Tab("Memory"):
|
| gr.Markdown("Store and recall memory, plus user/bot profile editing.")
|
|
|
| with gr.Row():
|
| mem_text = gr.Textbox(label="Text to store", lines=4)
|
| mem_tags = gr.Textbox(label="Tags (comma-separated)", value="manual")
|
| mem_importance = gr.Slider(0.0, 1.0, value=0.7, step=0.05, label="Importance")
|
| store_btn = gr.Button("Store memory")
|
| mem_store_out = gr.Textbox(label="Store result", lines=4)
|
|
|
| with gr.Row():
|
| mem_query = gr.Textbox(label="Recall query", lines=3)
|
| mem_top_k = gr.Slider(1, 20, value=5, step=1, label="Top K")
|
| recall_btn = gr.Button("Recall memory")
|
| mem_recall_out = gr.Textbox(label="Recall result", lines=10)
|
|
|
| gr.Markdown("Profiles")
|
| with gr.Row():
|
| user_profile_json = gr.Textbox(label="User profile JSON", lines=10)
|
| bot_profile_json = gr.Textbox(label="Bot profile JSON", lines=10)
|
| with gr.Row():
|
| refresh_profiles_btn = gr.Button("Refresh profiles")
|
| apply_profiles_btn = gr.Button("Apply profiles", variant="primary")
|
| profile_status = gr.Textbox(label="Profile status", lines=2)
|
|
|
| store_btn.click(
|
| store_memory,
|
| inputs=[mem_text, mem_tags, mem_importance],
|
| outputs=[mem_store_out],
|
| )
|
| recall_btn.click(
|
| recall_memory,
|
| inputs=[mem_query, mem_top_k],
|
| outputs=[mem_recall_out],
|
| )
|
| refresh_profiles_btn.click(
|
| refresh_profiles,
|
| inputs=[],
|
| outputs=[user_profile_json, bot_profile_json],
|
| )
|
| apply_profiles_btn.click(
|
| apply_profiles,
|
| inputs=[user_profile_json, bot_profile_json],
|
| outputs=[profile_status],
|
| )
|
|
|
| with gr.Tab("Web"):
|
| gr.Markdown("Direct web search through the embedded web module.")
|
| web_query = gr.Textbox(label="Search query", lines=3)
|
| web_deep = gr.Checkbox(value=False, label="Deep search")
|
| web_btn = gr.Button("Search", variant="primary")
|
| web_out = gr.Textbox(label="Results", lines=20)
|
| web_btn.click(web_search, inputs=[web_query, web_deep], outputs=[web_out])
|
|
|
| with gr.Tab("CodeBox"):
|
| gr.Markdown("Run code inside the embedded sandbox.")
|
| code_text = gr.Code(label="Python code", language="python", lines=18)
|
| with gr.Row():
|
| venv_id_in = gr.Textbox(label="Venv ID", value=DEFAULT_VENV_ID)
|
| timeout_in = gr.Slider(5, 600, value=120, step=5, label="Timeout (seconds)")
|
| max_ram_in = gr.Slider(256, 32768, value=4096, step=256, label="Max RAM (MB)")
|
| requirements_in = gr.Textbox(
|
| label="Requirements (one per line)",
|
| lines=5,
|
| placeholder="numpy\npandas\nrequests",
|
| )
|
| code_btn = gr.Button("Run code", variant="primary")
|
| code_out = gr.Textbox(label="Sandbox result", lines=20)
|
| code_btn.click(
|
| run_code,
|
| inputs=[code_text, venv_id_in, requirements_in, timeout_in, max_ram_in],
|
| outputs=[code_out],
|
| )
|
|
|
| with gr.Tab("System"):
|
| gr.Markdown("Warmup, status, load management, and checkpoint saving.")
|
|
|
| with gr.Row():
|
| warm_web = gr.Checkbox(value=False, label="Warm web")
|
| warm_vision = gr.Checkbox(value=False, label="Warm vision")
|
| warm_action = gr.Checkbox(value=False, label="Warm action")
|
| warm_btn = gr.Button("Warmup", variant="primary")
|
| warm_out = gr.Textbox(label="Warmup report", lines=10)
|
|
|
| status_btn = gr.Button("Refresh status")
|
| status_out = gr.Textbox(label="Status", lines=16)
|
|
|
| with gr.Row():
|
| expert_name = gr.Dropdown(
|
| choices=[
|
| "head_expert",
|
| "creative_expert",
|
| "code_expert",
|
| "logic_expert",
|
| "math_expert",
|
| "affect_expert",
|
| "role_expert",
|
| "vision_expert",
|
| "tool_expert",
|
| "translation_expert",
|
| "web_expert",
|
| "action_expert",
|
| ],
|
| value="head_expert",
|
| label="Expert to reload/unload",
|
| )
|
| save_path = gr.Textbox(label="Save path (blank = default)", value="")
|
| with gr.Row():
|
| reload_btn = gr.Button("Reload expert")
|
| unload_btn = gr.Button("Unload expert")
|
| unload_all_btn = gr.Button("Unload all")
|
| save_btn = gr.Button("Save checkpoint", variant="primary")
|
|
|
| reload_out = gr.Textbox(label="Reload result", lines=2)
|
| unload_out = gr.Textbox(label="Unload result", lines=2)
|
| save_out = gr.Textbox(label="Save result", lines=2)
|
|
|
| warm_btn.click(
|
| warmup_runner,
|
| inputs=[warm_web, warm_vision, warm_action],
|
| outputs=[warm_out],
|
| )
|
| status_btn.click(get_status, inputs=[], outputs=[status_out])
|
| reload_btn.click(reload_expert, inputs=[expert_name], outputs=[reload_out])
|
| unload_btn.click(unload_expert, inputs=[expert_name], outputs=[unload_out])
|
| unload_all_btn.click(unload_all, inputs=[], outputs=[unload_out])
|
| save_btn.click(save_checkpoint, inputs=[save_path], outputs=[save_out])
|
|
|
| demo.load(refresh_profiles, inputs=[], outputs=[user_profile_json, bot_profile_json])
|
|
|
| if __name__ == "__main__":
|
| demo.queue(default_concurrency_limit=1).launch() |