Spaces:
Paused
Paused
| import spaces | |
| import gradio as gr | |
| import subprocess, os, sys, time | |
| import urllib.request, urllib.error | |
| from fastapi import Request | |
| from fastapi.responses import Response, PlainTextResponse, HTMLResponse | |
| PORT = 9001 | |
| FM_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "file-manager") | |
| MODEL_ID = "Qwen/Qwen3-Coder-Next" | |
| # --- Launch the file manager as a background child process --- | |
| proc = {"p": None} | |
| def start_fm(): | |
| if proc["p"] and proc["p"].poll() is None: | |
| proc["p"].terminate() | |
| try: | |
| proc["p"].wait(timeout=5) | |
| except Exception: | |
| proc["p"].kill() | |
| proc["p"] = subprocess.Popen([sys.executable, "app.py"], cwd=FM_DIR, | |
| stdout=sys.stdout, stderr=sys.stderr) | |
| start_fm() | |
| # --- GPU info (runs INSIDE the gate; GPU only attached here) --- | |
| def gpu_report(): | |
| import torch | |
| lines = [f"torch.cuda.is_available(): {torch.cuda.is_available()}"] | |
| if torch.cuda.is_available(): | |
| props = torch.cuda.get_device_properties(0) | |
| lines.append(f"Device name: {torch.cuda.get_device_name(0)}") | |
| lines.append(f"Device count: {torch.cuda.device_count()}") | |
| lines.append(f"Total memory: {props.total_memory / 1024**3:.1f} GB") | |
| lines.append(f"Compute capability: {props.major}.{props.minor}") | |
| try: | |
| smi = subprocess.run(["nvidia-smi"], capture_output=True, text=True, timeout=30) | |
| lines.append("\n--- nvidia-smi ---\n" + smi.stdout + smi.stderr) | |
| except Exception as e: | |
| lines.append(f"nvidia-smi error: {e}") | |
| return "\n".join(lines) | |
| # --- LLM (loads + generates INSIDE the gate) --- | |
| _state = {"model": None, "tok": None} | |
| def run_llm(prompt: str): | |
| import torch | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| if _state["model"] is None: | |
| _state["tok"] = AutoTokenizer.from_pretrained(MODEL_ID) | |
| _state["model"] = AutoModelForCausalLM.from_pretrained( | |
| MODEL_ID, torch_dtype=torch.float16, device_map="cuda") | |
| tok, model = _state["tok"], _state["model"] | |
| msgs = [{"role": "user", "content": prompt}] | |
| text = tok.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True) | |
| inputs = tok(text, return_tensors="pt").to("cuda") | |
| out = model.generate(**inputs, max_new_tokens=256) | |
| reply = tok.decode(out[0][inputs.input_ids.shape[1]:], skip_special_tokens=True) | |
| return f"[GPU: {torch.cuda.get_device_name(0)}]\n\n{reply}" | |
| # --- Gradio owns the ONLY public port 7860 --- | |
| with gr.Blocks() as demo: | |
| gr.Markdown("Self-editing computer. Terminal + file manager at /fm") | |
| with gr.Row(): | |
| prompt = gr.Textbox(label="Ask the LLM") | |
| out = gr.Textbox(label="Reply") | |
| gr.Button("Run LLM").click(run_llm, inputs=prompt, outputs=out) | |
| gpu_out = gr.Textbox(label="GPU") | |
| gr.Button("GPU info").click(gpu_report, outputs=gpu_out) | |
| demo.queue() | |
| demo.launch(server_name="0.0.0.0", server_port=7860, | |
| ssr_mode=False, prevent_thread_lock=True) | |
| app = demo.app | |
| # --- Hot-reload: respawn ONLY the file-manager process, no Space rebuild --- | |
| def reload_fm(): | |
| start_fm() | |
| time.sleep(1) | |
| return HTMLResponse('Reloaded file-manager process. <a href="/fm/">back to /fm</a>') | |
| def gpuinfo(): | |
| return PlainTextResponse(gpu_report()) | |
| def llm(q: str = "Say hello and tell me which GPU you are running on."): | |
| return PlainTextResponse(run_llm(q)) | |
| # --- Proxy: /fm/... -> 127.0.0.1:9001/fm/... (method + body forwarded) --- | |
| async def proxy(request: Request, path: str = ""): | |
| url = f"http://127.0.0.1:{PORT}/fm/{path}" | |
| if request.url.query: | |
| url += "?" + request.url.query | |
| body = await request.body() | |
| fwd = {k: v for k, v in request.headers.items() if k.lower() != "host"} | |
| req = urllib.request.Request(url, data=body or None, method=request.method, headers=fwd) | |
| try: | |
| with urllib.request.urlopen(req) as r: | |
| data, status, headers = r.read(), r.status, r.headers | |
| except urllib.error.HTTPError as e: | |
| data, status, headers = e.read(), e.code, e.headers | |
| except urllib.error.URLError: | |
| return Response(content=b"file-manager not up yet on 9001", status_code=502) | |
| out = {k: v for k, v in headers.items() | |
| if k.lower() in ("set-cookie", "location", "content-type")} | |
| return Response(content=data, status_code=status, headers=out) | |
| while True: | |
| time.sleep(3600) |