Nvidia NIM Models & Health Status
Checked every 15 minsSystem Activity Logs
Visual Workspace IDE
Welcome to Claude Code Workspace Explorer.
Select a file from the sidebar explorer on the left to read its code contents in real-time.
""" Claude Code Backend ā Agentic coding backend powered by NVIDIA NIM models. Exposes an OpenAI-compatible /v1/chat/completions endpoint with built-in tools for file operations and bash execution. Architecture: Space 1 (better-chatbot) --> this backend --> NVIDIA NIM API The agentic loop: 1. Receive user message from Space 1 2. Send to NIM model with tool definitions 3. If model returns tool_calls, execute them and loop 4. If model returns text, stream it back to Space 1 5. Persist conversation in Postgres """ import os import shutil import threading import json import uuid import subprocess import asyncio import time import re import collections from pathlib import Path from typing import AsyncIterator, Optional, List, Dict, Any from pydantic import BaseModel from fastapi import FastAPI, Request, Header, HTTPException from fastapi.responses import StreamingResponse, JSONResponse, HTMLResponse from fastapi.middleware.cors import CORSMiddleware from openai import AsyncOpenAI import anyio import asyncpg # --------------------------------------------------------------------------- # Globals & Activity Logs # --------------------------------------------------------------------------- activity_logs = collections.deque(maxlen=100) MODEL_STATUSES = {} ACTIVE_SESSIONS = set() def log_activity(msg: str): timestamp = time.strftime("%H:%M:%S") log_line = f"[{timestamp}] {msg}" activity_logs.append(log_line) print(log_line) # --------------------------------------------------------------------------- # Configuration # --------------------------------------------------------------------------- NIM_API_KEY = os.environ.get("NVIDIA_NIM_API_KEY", "") BACKEND_API_KEY = os.environ.get("BACKEND_API_KEY", "") DATABASE_URL = os.environ.get("DATABASE_URL", "") WORKSPACE_DIR = os.environ.get("WORKSPACE_DIR", "/tmp/workspace") BACKUP_GIT_REPO = os.environ.get("BACKUP_GIT_REPO", "") MAX_TOOL_ROUNDS = int(os.environ.get("MAX_TOOL_ROUNDS", "10")) # NIM models that reliably support tool/function calling TOOL_CAPABLE_MODELS = { "nvidia/nemotron-3-ultra-550b-a55b": "Nemotron 3 Ultra 550B (Agentic)", "z-ai/glm-5.1": "GLM 5.1 (Agentic)", "moonshotai/kimi-k2.6": "Kimi K2.6 (Agentic)", "minimaxai/minimax-m3": "MiniMax M3 (Agentic)", "stepfun-ai/step-3.7-flash": "Step 3.7 Flash (Agentic)", "minimaxai/minimax-m2.7": "MiniMax M2.7 (Agentic)", "meta/llama-3.1-70b-instruct": "Llama 3.1 70B (Agentic)", "meta/llama-3.1-405b-instruct": "Llama 3.1 405B (Agentic)", "qwen/qwen2.5-coder-32b-instruct": "Qwen 2.5 Coder 32B (Agentic)", "nvidia/llama-3.1-nemotron-70b-instruct": "Nemotron 70B (Agentic)", "meta/llama-3.3-70b-instruct": "Llama 3.3 70B (Agentic)", } # All models (tool-capable get agentic mode, others get plain chat) ALL_MODELS = { **TOOL_CAPABLE_MODELS, "deepseek-ai/deepseek-r1": "DeepSeek R1 (Chat only)", "mistralai/mistral-large-2-instruct": "Mistral Large 2 (Chat only)", } RECOMMENDED_MODEL = "nvidia/llama-3.1-nemotron-70b-instruct" # Ensure workspace exists Path(WORKSPACE_DIR).mkdir(parents=True, exist_ok=True) # --------------------------------------------------------------------------- # NIM Client # --------------------------------------------------------------------------- nim_client = AsyncOpenAI( base_url="https://integrate.api.nvidia.com/v1", api_key=NIM_API_KEY, ) # --------------------------------------------------------------------------- # Rate Limiting & Multi-Provider Setup # --------------------------------------------------------------------------- MISTRAL_API_KEY = os.environ.get("MISTRAL_API_KEY", "") mistral_client = AsyncOpenAI( base_url="https://api.mistral.ai/v1", api_key=MISTRAL_API_KEY if MISTRAL_API_KEY else "dummy_key", ) if MISTRAL_API_KEY else None class MultiProviderRateLimiter: def __init__(self): self.nim_limit = 40 self.nim_window = 60 self.nim_calls = [] self.mistral_last_call = 0.0 self.lock = asyncio.Lock() async def wait_for_mistral(self): async with self.lock: now = time.time() elapsed = now - self.mistral_last_call if elapsed < 1.0: await asyncio.sleep(1.0 - elapsed) self.mistral_last_call = time.time() async def wait_for_nim(self): async with self.lock: now = time.time() self.nim_calls = [t for t in self.nim_calls if now - t < self.nim_window] if len(self.nim_calls) >= self.nim_limit - 2: sleep_time = self.nim_window - (now - self.nim_calls[0]) print(f"[RateLimiter] Approaching NIM rate limit (40 RPM). Sleeping {sleep_time:.2f}s...") await asyncio.sleep(sleep_time) self.nim_calls.append(time.time()) rate_limiter = MultiProviderRateLimiter() # --------------------------------------------------------------------------- # Tool Definitions (OpenAI function calling format) # --------------------------------------------------------------------------- TOOLS = [ { "type": "function", "function": { "name": "read_file", "description": "Read the contents of a file. Use this to inspect existing code, configs, or any text file.", "parameters": { "type": "object", "properties": { "path": { "type": "string", "description": "Relative path to the file from the workspace root" } }, "required": ["path"] } } }, { "type": "function", "function": { "name": "write_file", "description": "Write content to a file. Creates the file if it doesn't exist, overwrites if it does. Creates parent directories automatically.", "parameters": { "type": "object", "properties": { "path": { "type": "string", "description": "Relative path to the file from the workspace root" }, "content": { "type": "string", "description": "The full content to write to the file" } }, "required": ["path", "content"] } } }, { "type": "function", "function": { "name": "run_bash", "description": "Execute a bash command in the workspace directory. Use for installing packages, running scripts, git operations, etc. Commands run with a 30 second timeout.", "parameters": { "type": "object", "properties": { "command": { "type": "string", "description": "The bash command to execute" } }, "required": ["command"] } } }, { "type": "function", "function": { "name": "list_directory", "description": "List files and directories in a given path. Shows file sizes and directory markers.", "parameters": { "type": "object", "properties": { "path": { "type": "string", "description": "Relative path to the directory from workspace root. Use '.' for the workspace root." } }, "required": ["path"] } } }, { "type": "function", "function": { "name": "grep_search", "description": "Search for a pattern in files within the workspace. Returns matching lines with file paths and line numbers.", "parameters": { "type": "object", "properties": { "pattern": { "type": "string", "description": "The search pattern (supports basic regex)" }, "path": { "type": "string", "description": "Directory or file to search in, relative to workspace root. Defaults to '.'", } }, "required": ["pattern"] } } }, ] # --------------------------------------------------------------------------- # Tool Execution # --------------------------------------------------------------------------- def _safe_path(rel_path: str) -> Path: """Resolve a relative path safely within the workspace.""" workspace = Path(WORKSPACE_DIR).resolve() target = (workspace / rel_path).resolve() # Prevent path traversal if not str(target).startswith(str(workspace)): raise ValueError(f"Path traversal detected: {rel_path}") return target def repair_arguments(func_name: str, args: dict) -> tuple[dict, list[str]]: notes = [] repaired_args = dict(args) # 1. Nesting extraction (e.g. {"path": {"path": "file.txt"}}) for key in list(repaired_args.keys()): val = repaired_args[key] if isinstance(val, dict) and key in val: repaired_args[key] = val[key] notes.append(f"Flattened nested parameter '{key}'") # 2. Markdown stripping from bash command if func_name == "run_bash" and "command" in repaired_args: cmd = repaired_args["command"] if isinstance(cmd, str): pattern = r"```(?:bash)?\s*(.*?)\s*```" match = re.search(pattern, cmd, re.DOTALL) if match: repaired_args["command"] = match.group(1).strip() notes.append("Stripped markdown code blocks from bash command") # 3. Stringified array conversion for key, val in repaired_args.items(): if isinstance(val, str) and val.strip().startswith("[") and val.strip().endswith("]"): try: parsed_arr = json.loads(val) if isinstance(parsed_arr, list): repaired_args[key] = parsed_arr notes.append(f"Converted stringified array for parameter '{key}' to native array") except: pass # 4. Optional empty objects replacing Null for key in list(repaired_args.keys()): if repaired_args[key] == {}: repaired_args[key] = None notes.append(f"Replaced empty object for parameter '{key}' with null") return repaired_args, notes async def execute_tool(name: str, arguments: dict) -> str: """Execute a tool and return its output as a string asynchronously.""" try: if name == "read_file": path = _safe_path(arguments["path"]) if not path.exists(): return f"Error: File not found: {arguments['path']}" if not path.is_file(): return f"Error: Not a file: {arguments['path']}" content = path.read_text(encoding="utf-8", errors="replace") if len(content) > 50000: return content[:50000] + f"\n\n[Truncated ā file is {len(content)} chars]" return content elif name == "write_file": path = _safe_path(arguments["path"]) path.parent.mkdir(parents=True, exist_ok=True) path.write_text(arguments["content"], encoding="utf-8") return f"Successfully wrote {len(arguments['content'])} chars to {arguments['path']}" elif name == "run_bash": command = arguments["command"] # Safety: block dangerous commands blocked = ["rm -rf /", "mkfs", "dd if=", ":(){", "fork bomb"] if any(b in command.lower() for b in blocked): return "Error: Command blocked for safety reasons" # ASYNC SUBPROCESS - This prevents the FastAPI server from freezing! process = await asyncio.create_subprocess_shell( command, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE, cwd=WORKSPACE_DIR, env={**os.environ, "HOME": "/tmp", "PATH": os.environ.get("PATH", "/usr/local/bin:/usr/bin:/bin")}, ) try: stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=30) output = "" if stdout: output += stdout.decode('utf-8', errors='replace') if stderr: output += ("\n" if output else "") + f"[stderr] {stderr.decode('utf-8', errors='replace')}" if process.returncode != 0: output += f"\n[exit code: {process.returncode}]" if not output: output = "[command completed with no output]" except asyncio.TimeoutError: try: process.kill() except Exception: pass await process.communicate() return "Error: Command timed out after 30 seconds" if len(output) > 20000: output = output[:20000] + f"\n\n[Truncated ā output is {len(output)} chars]" return output elif name == "list_directory": path = _safe_path(arguments.get("path", ".")) if not path.exists(): return f"Error: Directory not found: {arguments.get('path', '.')}" if not path.is_dir(): return f"Error: Not a directory: {arguments.get('path', '.')}" entries = [] for item in sorted(path.iterdir()): if item.is_dir(): entries.append(f" š {item.name}/") else: size = item.stat().st_size if size < 1024: size_str = f"{size}B" elif size < 1024 * 1024: size_str = f"{size/1024:.1f}KB" else: size_str = f"{size/(1024*1024):.1f}MB" entries.append(f" š {item.name} ({size_str})") return f"Contents of {arguments.get('path', '.')}:\n" + "\n".join(entries) if entries else "Empty directory" elif name == "grep_search": pattern = arguments["pattern"] search_path = arguments.get("path", ".") path = _safe_path(search_path) # Escape single quotes in pattern for safety escaped_pattern = pattern.replace("'", "'\\''") process = await asyncio.create_subprocess_shell( f"grep -rn --include=* '{escaped_pattern}' '{path}'", stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE, cwd=WORKSPACE_DIR, ) try: stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=10) output = stdout.decode('utf-8', errors='replace') if stdout else "No matches found" except asyncio.TimeoutError: try: process.kill() except Exception: pass await process.communicate() return "Error: Grep search timed out" if len(output) > 10000: output = output[:10000] + "\n\n[Truncated]" return output else: return f"Error: Unknown tool: {name}" except Exception as e: return f"Error executing {name}: {str(e)}" # --------------------------------------------------------------------------- # Database (Session Persistence) # --------------------------------------------------------------------------- db_pool: Optional[asyncpg.Pool] = None async def init_db(): """Initialize database connection pool and create tables.""" global db_pool if not DATABASE_URL: return try: db_pool = await asyncpg.create_pool( DATABASE_URL, ssl="require", min_size=1, max_size=3, max_inactive_connection_lifetime=300 ) async with db_pool.acquire() as conn: await conn.execute(""" CREATE TABLE IF NOT EXISTS agent_session_entries ( id BIGSERIAL PRIMARY KEY, project_key TEXT NOT NULL, session_id TEXT NOT NULL, subpath TEXT, entry JSONB NOT NULL, created_at TIMESTAMPTZ NOT NULL DEFAULT now() ); CREATE INDEX IF NOT EXISTS idx_session_key ON agent_session_entries (project_key, session_id, subpath, id); CREATE INDEX IF NOT EXISTS idx_project_session ON agent_session_entries (project_key, session_id); CREATE TABLE IF NOT EXISTS eternity_system_state ( id SERIAL PRIMARY KEY, goal TEXT NOT NULL, deadline TIMESTAMPTZ NOT NULL, current_mode VARCHAR(20) NOT NULL DEFAULT 'build', roadmap JSONB DEFAULT '[]', latest_brief TEXT, created_at TIMESTAMPTZ NOT NULL DEFAULT now() ); """) except Exception as e: print(f"[DB] Warning: Could not initialize database: {e}") db_pool = None async def save_message(session_id: str, role: str, content: str = None, tool_calls: list = None, tool_call_id: str = None): """Save a message to the session store using the unified schema.""" if not db_pool: return msg = {"role": role} if content is not None: msg["content"] = content if tool_calls: msg["tool_calls"] = tool_calls if tool_call_id: msg["tool_call_id"] = tool_call_id try: async with db_pool.acquire() as conn: await conn.execute( "INSERT INTO agent_session_entries (project_key, session_id, subpath, entry) VALUES ($1, $2, $3, $4)", "fastapi-completions", session_id, None, json.dumps(msg) ) except Exception as e: print(f"[DB] Warning: Could not save message: {e}") async def load_session(session_id: str) -> list: """Load conversation history from the session store using the unified schema.""" if not db_pool: return [] try: async with db_pool.acquire() as conn: rows = await conn.fetch( "SELECT entry FROM agent_session_entries WHERE project_key = $1 AND session_id = $2 AND subpath IS NOT DISTINCT FROM $3 ORDER BY id", "fastapi-completions", session_id, None ) return [json.loads(row["entry"]) for row in rows] except Exception as e: print(f"[DB] Warning: Could not load session: {e}") return [] # --------------------------------------------------------------------------- # SSE Chunk Formatting (OpenAI delta format) # --------------------------------------------------------------------------- def make_chunk(request_id: str, model: str, content: str = "", finish_reason: str = None) -> str: """Create an OpenAI-compatible SSE chunk.""" delta = {} if content: delta["content"] = content if finish_reason and not content: delta = {} chunk = { "id": f"chatcmpl-{request_id}", "object": "chat.completion.chunk", "created": int(time.time()), "model": model, "choices": [{ "index": 0, "delta": delta, "finish_reason": finish_reason, }], } return f"data: {json.dumps(chunk)}\n\n" # --------------------------------------------------------------------------- # FastAPI Application # --------------------------------------------------------------------------- app = FastAPI(title="Claude Code Backend", version="1.0.0") app.add_middleware( CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"], ) def auth(authorization: str = None): """Verify bearer token.""" if not BACKEND_API_KEY: return # No auth configured expected = f"Bearer {BACKEND_API_KEY}" if authorization != expected: raise HTTPException(status_code=401, detail="Unauthorized") async def check_models_health(): global RECOMMENDED_MODEL # Test only the unstable frontier models (the rest). # The stable ones (Step 3.7 Flash, Nemotron 3 Ultra, Qwen 2.5 Coder) are always free/working. models_to_test = [ "moonshotai/kimi-k2.6", "z-ai/glm-5.1", "minimaxai/minimax-m3", "minimaxai/minimax-m2.7", "meta/llama-3.1-405b-instruct", ] best_model = None best_latency = 999.0 # Mark stable models as permanently ONLINE in the status map stable_models = [ "stepfun-ai/step-3.7-flash", "nvidia/nemotron-3-ultra-550b-a55b", "qwen/qwen2.5-coder-32b-instruct" ] for model in stable_models: MODEL_STATUSES[model] = {"status": "ONLINE (Stable)", "latency": "Fast", "raw_latency": 0.1} log_activity("Periodic health check started: verifying unstable frontier NIM models...") for model in models_to_test: start_time = time.time() try: # Send a fast test prompt async with anyio.fail_after(15.0): # 15 seconds max timeout await nim_client.chat.completions.create( model=model, messages=[{"role": "user", "content": "1+1="}], max_tokens=3, ) latency = time.time() - start_time MODEL_STATUSES[model] = {"status": "ONLINE", "latency": f"{latency:.2f}s", "raw_latency": latency} log_activity(f"Model checked: {model} is ONLINE ({latency:.2f}s)") # Choose the fastest online unstable model if latency < best_latency: best_latency = latency best_model = model except Exception as e: MODEL_STATUSES[model] = {"status": "OFFLINE", "latency": "N/A", "raw_latency": 999.0} log_activity(f"Model checked: {model} is OFFLINE / TIMEOUT: {e}") if best_model: RECOMMENDED_MODEL = best_model log_activity(f"Best frontier model selected: {RECOMMENDED_MODEL} ({best_latency:.2f}s)") else: # Fallback to the stable Step 3.7 Flash if all frontier models are offline/throttled RECOMMENDED_MODEL = "stepfun-ai/step-3.7-flash" log_activity(f"All frontier models offline. Falling back to stable recommended model: {RECOMMENDED_MODEL}") async def periodic_health_check_loop(): # Wait 10 seconds after startup before the first check to let the space boot fully await asyncio.sleep(10) while True: try: await check_models_health() except Exception as e: log_activity(f"Health check loop error: {e}") await asyncio.sleep(900) # every 15 minutes (reduce frequency to save quota) @app.on_event("startup") async def startup(): await init_db() Path(WORKSPACE_DIR).mkdir(parents=True, exist_ok=True) # Initialize statuses for all models for model_id, display_name in ALL_MODELS.items(): MODEL_STATUSES[model_id] = {"status": "UNCHECKED", "latency": "N/A", "raw_latency": 999.0} # Start background health checking asyncio.create_task(periodic_health_check_loop()) log_activity(f"FastAPI backend started. Workspace: {WORKSPACE_DIR}") # --------------------------------------------------------------------------- # /v1/chat/completions ā Main endpoint # --------------------------------------------------------------------------- AGENTIC_SYSTEM_PROMPT = """You are an expert coding assistant with access to tools for file operations and command execution. When the user asks you to create, edit, or debug code: 1. Use `list_directory` and `read_file` to understand the current state 2. Use `write_file` to create or modify files 3. Use `run_bash` to execute commands (install packages, run scripts, test code) 4. Use `grep_search` to find patterns in code IMPORTANT RULES: - Always use tools to take action. Do NOT just describe what to do ā actually DO it. - After writing code, run it to verify it works. - If a command fails, read the error and fix it. - Work in the /tmp/workspace directory. - Be concise in your explanations, but thorough in your tool usage. """ def compact_history(messages: list) -> list: """ Compact conversation history to prevent context window overflow. Replaces massive tool call outputs with concise summaries if history is long. """ # Only compact if messages count exceeds 15 (to maintain normal conversation) if len(messages) <= 15: return messages compacted = [] # Always keep the system prompt (typically the first message) if messages and messages[0].get("role") == "system": compacted.append(messages[0]) start_idx = 1 else: start_idx = 0 # Keep the last 4 messages exactly as they are to preserve immediate context recent_count = 4 mid_messages = messages[start_idx:-recent_count] recent_messages = messages[-recent_count:] for msg in mid_messages: role = msg.get("role") content = msg.get("content") or "" if role == "tool": # Compress massive tool outputs (like bash stdout or file reads) if len(content) > 1000: summary = f"[Tool output compacted: {content[:200]}... (Total {len(content)} chars truncated for context preservation)]" compacted.append({ "role": "tool", "tool_call_id": msg.get("tool_call_id"), "content": summary }) continue elif role == "assistant" and msg.get("tool_calls"): # Keep tool calls metadata so the model's message-tool call mapping DAG doesn't break pass # Keep general messages, but truncate if they are too long if len(content) > 2000: msg = dict(msg) msg["content"] = content[:2000] + "\n[Content truncated for compaction]" compacted.append(msg) compacted.extend(recent_messages) log_activity(f"[Auto-Compaction] Compressed message history from {len(messages)} down to {len(compacted)}") return compacted @app.post("/v1/chat/completions") async def chat_completions(request: Request, authorization: str = Header(None)): auth(authorization) body = await request.json() requested_model = body.get("model", "meta/llama-3.1-70b-instruct") messages = body.get("messages", []) stream = body.get("stream", False) session_id = body.get("session_id") or str(uuid.uuid4()) is_agentic = requested_model in TOOL_CAPABLE_MODELS request_id = str(uuid.uuid4())[:8] ACTIVE_SESSIONS.add(session_id) log_activity(f"Session [{session_id[:6]}] connected. Model: {requested_model}") # Build message history final_messages = [] # Add agentic system prompt for tool-capable models if is_agentic: # Check if there's already a system message has_system = any(m.get("role") == "system" for m in messages) if has_system: # Prepend agentic prompt to existing system message for m in messages: if m["role"] == "system": final_messages.append({ "role": "system", "content": AGENTIC_SYSTEM_PROMPT + "\n\nAdditional instructions:\n" + m["content"] }) else: final_messages.append(m) else: final_messages.append({"role": "system", "content": AGENTIC_SYSTEM_PROMPT}) final_messages.extend(messages) else: final_messages = list(messages) # Perform auto-compaction before executing agent loops final_messages = compact_history(final_messages) # Save the user's message to DB user_msg = next((m for m in reversed(messages) if m.get("role") == "user"), None) if user_msg: await save_message(session_id, "user", user_msg.get("content", "")) if not stream: # Non-streaming: simple completion try: kwargs = {"model": requested_model, "messages": final_messages} if is_agentic: kwargs["tools"] = TOOLS kwargs["tool_choice"] = "auto" async with completions_semaphore: response = await nim_client.chat.completions.create(**kwargs) content = response.choices[0].message.content or "" await save_message(session_id, "assistant", content) ACTIVE_SESSIONS.discard(session_id) log_activity(f"Session [{session_id[:6]}] finished (non-streaming)") return JSONResponse({ "id": f"chatcmpl-{request_id}", "object": "chat.completion", "created": int(time.time()), "model": requested_model, "choices": [{"index": 0, "message": {"role": "assistant", "content": content}, "finish_reason": "stop"}], }) except Exception as e: ACTIVE_SESSIONS.discard(session_id) return JSONResponse({"error": {"message": str(e), "type": "internal_error"}}, status_code=500) # Streaming + agentic loop async def generate() -> AsyncIterator[str]: nonlocal final_messages async with completions_semaphore: try: for round_num in range(MAX_TOOL_ROUNDS + 1): # Perform auto-compaction before calling NIM API final_messages = compact_history(final_messages) kwargs = {"model": requested_model, "messages": final_messages, "stream": True} if is_agentic: kwargs["tools"] = TOOLS kwargs["tool_choice"] = "auto" # Collect streamed response full_content = "" tool_calls_raw = {} # index -> {id, name, arguments_str} async for chunk in await nim_client.chat.completions.create(**kwargs): choice = chunk.choices[0] if chunk.choices else None if not choice: continue delta = choice.delta # Stream text content to client if delta and delta.content: full_content += delta.content yield make_chunk(request_id, requested_model, delta.content) # Collect tool calls if delta and delta.tool_calls: for tc in delta.tool_calls: idx = tc.index if idx not in tool_calls_raw: tool_calls_raw[idx] = { "id": tc.id or f"call_{uuid.uuid4().hex[:8]}", "name": tc.function.name if tc.function and tc.function.name else "", "arguments": "" } if tc.function and tc.function.name: tool_calls_raw[idx]["name"] = tc.function.name if tc.id: tool_calls_raw[idx]["id"] = tc.id if tc.function and tc.function.arguments: tool_calls_raw[idx]["arguments"] += tc.function.arguments # Check for finish if choice.finish_reason == "stop": break if choice.finish_reason == "tool_calls": break # If no tool calls, we're done if not tool_calls_raw: await save_message(session_id, "assistant", full_content) yield make_chunk(request_id, requested_model, finish_reason="stop") yield "data: [DONE]\n\n" return # Execute tool calls tool_calls_list = [] for idx in sorted(tool_calls_raw.keys()): tc = tool_calls_raw[idx] tool_calls_list.append({ "id": tc["id"], "type": "function", "function": {"name": tc["name"], "arguments": tc["arguments"]} }) # Add assistant message with tool calls to history assistant_msg = {"role": "assistant", "content": full_content or None, "tool_calls": tool_calls_list} final_messages.append(assistant_msg) # Execute each tool and add results for tc in tool_calls_list: func_name = tc["function"]["name"] raw_args_str = tc["function"]["arguments"] try: func_args = json.loads(raw_args_str) except json.JSONDecodeError: # Attempt raw JSON repair repaired_str = raw_args_str.strip() if not repaired_str.startswith("{"): repaired_str = "{" + repaired_str if not repaired_str.endswith("}"): repaired_str = repaired_str + "}" try: func_args = json.loads(repaired_str) log_activity(f"Auto-fixed invalid JSON string for tool: {func_name}") except: func_args = {} # Perform semantic repairs repaired_args, repair_notes = repair_arguments(func_name, func_args) # Log activity log_activity(f"Tool execution: {func_name} args={repaired_args}") if repair_notes: for note in repair_notes: log_activity(f"[Tool Repair] {note}") # Show tool execution to user yield make_chunk(request_id, requested_model, f"\n\nš§ **{func_name}**") if repair_notes: yield make_chunk(request_id, requested_model, " *(Auto-Repaired)*") if func_name == "run_bash" and "command" in repaired_args: yield make_chunk(request_id, requested_model, f": `{repaired_args['command']}`\n") elif func_name == "read_file" and "path" in repaired_args: yield make_chunk(request_id, requested_model, f": `{repaired_args['path']}`\n") elif func_name == "write_file" and "path" in repaired_args: yield make_chunk(request_id, requested_model, f": `{repaired_args['path']}`\n") elif func_name == "list_directory": yield make_chunk(request_id, requested_model, f": `{repaired_args.get('path', '.')}`\n") elif func_name == "grep_search": yield make_chunk(request_id, requested_model, f": `{repaired_args.get('pattern', '')}`\n") else: yield make_chunk(request_id, requested_model, "\n") # Execute the tool result = await execute_tool(func_name, repaired_args) # Append teaching note if repaired if repair_notes: result += f"\n\n[SYSTEM REPAIR NOTE: The harness automatically fixed formatting issues: {', '.join(repair_notes)}. Please strictly follow the tool's JSON schema in subsequent calls without these wrapping/formatting errors.]" # Show truncated result to user preview = result[:500] + ("..." if len(result) > 500 else "") yield make_chunk(request_id, requested_model, f"```\n{preview}\n```\n") # Add tool result to message history final_messages.append({ "role": "tool", "tool_call_id": tc["id"], "content": result, }) await save_message(session_id, "tool", result, tool_call_id=tc["id"]) # Continue the agentic loop (model processes tool results) # If we hit max rounds, finish yield make_chunk(request_id, requested_model, "\n\nā ļø Reached maximum tool call rounds.") yield make_chunk(request_id, requested_model, finish_reason="stop") yield "data: [DONE]\n\n" except Exception as e: error_msg = f"\n\nā Error: {str(e)}" yield make_chunk(request_id, requested_model, error_msg) yield make_chunk(request_id, requested_model, finish_reason="stop") yield "data: [DONE]\n\n" return StreamingResponse( generate(), media_type="text/event-stream", headers={ "Cache-Control": "no-cache", "X-Accel-Buffering": "no", "Connection": "keep-alive", }, ) # --------------------------------------------------------------------------- # /v1/models ā Model listing # --------------------------------------------------------------------------- @app.get("/v1/models") async def list_models(authorization: str = Header(None)): auth(authorization) models = [] for model_id, display_name in ALL_MODELS.items(): models.append({ "id": model_id, "object": "model", "created": 1700000000, "owned_by": "nvidia-nim", "permission": [], "root": model_id, "parent": None, }) return {"object": "list", "data": models} # --------------------------------------------------------------------------- # /health ā Health check # --------------------------------------------------------------------------- @app.get("/api/workspace/tree") async def get_workspace_tree(): def build_tree(current_path: Path, relative_to: Path) -> dict: name = current_path.name try: rel_path = str(current_path.relative_to(relative_to)).replace("\\", "/") except ValueError: rel_path = "" if rel_path == ".": rel_path = "" if current_path.is_dir(): children = [] try: for child in sorted(current_path.iterdir(), key=lambda x: (not x.is_dir(), x.name)): if child.name in [".git", "node_modules", ".next", "__pycache__", ".agents", ".gemini"]: continue children.append(build_tree(child, relative_to)) except Exception: pass return { "name": name or "workspace", "path": rel_path, "type": "directory", "children": children } else: return { "name": name, "path": rel_path, "type": "file", "size": current_path.stat().st_size if current_path.exists() else 0 } try: w_path = Path(WORKSPACE_DIR).resolve() if not w_path.exists(): w_path.mkdir(parents=True, exist_ok=True) return build_tree(w_path, w_path) except Exception as e: return {"error": str(e)} @app.get("/api/workspace/file") async def get_workspace_file(path: str): try: safe_p = _safe_path(path) if not safe_p.exists() or not safe_p.is_file(): raise HTTPException(status_code=404, detail="File not found") content = safe_p.read_text(encoding="utf-8", errors="replace") return {"path": path, "content": content} except Exception as e: raise HTTPException(status_code=500, detail=str(e)) # --------------------------------------------------------------------------- # Dashboard and Status API # --------------------------------------------------------------------------- DASHBOARD_HTML = """
Welcome to Claude Code Workspace Explorer.
Select a file from the sidebar explorer on the left to read its code contents in real-time.