File size: 4,596 Bytes
6f3d5da
 
 
 
 
fbc7990
6f3d5da
f0c48a6
 
9b9a92c
6f3d5da
fbc7990
f0c48a6
 
 
 
 
 
 
 
 
 
6f3d5da
f0c48a6
 
028a8f1
6f3d5da
b55ae70
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fbc7990
b55ae70
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6f3d5da
fbc7990
6f3d5da
fbc7990
b55ae70
 
 
 
 
 
6f3d5da
 
f0c48a6
 
6f3d5da
fbc7990
6f3d5da
fbc7990
f0c48a6
 
 
 
 
6f3d5da
b55ae70
 
 
 
 
 
 
 
fbc7990
f0c48a6
 
 
 
6f3d5da
 
 
f0c48a6
 
6f3d5da
fbc7990
 
6f3d5da
fbc7990
6f3d5da
f0c48a6
fbc7990
 
 
6f3d5da
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
import spaces
import gradio as gr
import subprocess, os, sys, time
import urllib.request, urllib.error
from fastapi import Request
from fastapi.responses import Response, PlainTextResponse, HTMLResponse

PORT = 9001
FM_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "file-manager")
MODEL_ID = "Qwen/Qwen3-Coder-Next"

# --- Launch the file manager as a background child process ---
proc = {"p": None}
def start_fm():
    if proc["p"] and proc["p"].poll() is None:
        proc["p"].terminate()
        try:
            proc["p"].wait(timeout=5)
        except Exception:
            proc["p"].kill()
    proc["p"] = subprocess.Popen([sys.executable, "app.py"], cwd=FM_DIR,
                                 stdout=sys.stdout, stderr=sys.stderr)

start_fm()

# --- GPU info (runs INSIDE the gate; GPU only attached here) ---
@spaces.GPU
def gpu_report():
    import torch
    lines = [f"torch.cuda.is_available(): {torch.cuda.is_available()}"]
    if torch.cuda.is_available():
        props = torch.cuda.get_device_properties(0)
        lines.append(f"Device name: {torch.cuda.get_device_name(0)}")
        lines.append(f"Device count: {torch.cuda.device_count()}")
        lines.append(f"Total memory: {props.total_memory / 1024**3:.1f} GB")
        lines.append(f"Compute capability: {props.major}.{props.minor}")
    try:
        smi = subprocess.run(["nvidia-smi"], capture_output=True, text=True, timeout=30)
        lines.append("\n--- nvidia-smi ---\n" + smi.stdout + smi.stderr)
    except Exception as e:
        lines.append(f"nvidia-smi error: {e}")
    return "\n".join(lines)

# --- LLM (loads + generates INSIDE the gate) ---
_state = {"model": None, "tok": None}

@spaces.GPU
def run_llm(prompt: str):
    import torch
    from transformers import AutoModelForCausalLM, AutoTokenizer
    if _state["model"] is None:
        _state["tok"] = AutoTokenizer.from_pretrained(MODEL_ID)
        _state["model"] = AutoModelForCausalLM.from_pretrained(
            MODEL_ID, torch_dtype=torch.float16, device_map="cuda")
    tok, model = _state["tok"], _state["model"]
    msgs = [{"role": "user", "content": prompt}]
    text = tok.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
    inputs = tok(text, return_tensors="pt").to("cuda")
    out = model.generate(**inputs, max_new_tokens=256)
    reply = tok.decode(out[0][inputs.input_ids.shape[1]:], skip_special_tokens=True)
    return f"[GPU: {torch.cuda.get_device_name(0)}]\n\n{reply}"

# --- Gradio owns the ONLY public port 7860 ---
with gr.Blocks() as demo:
    gr.Markdown("Self-editing computer. Terminal + file manager at /fm")
    with gr.Row():
        prompt = gr.Textbox(label="Ask the LLM")
        out = gr.Textbox(label="Reply")
    gr.Button("Run LLM").click(run_llm, inputs=prompt, outputs=out)
    gpu_out = gr.Textbox(label="GPU")
    gr.Button("GPU info").click(gpu_report, outputs=gpu_out)

demo.queue()
demo.launch(server_name="0.0.0.0", server_port=7860,
            ssr_mode=False, prevent_thread_lock=True)

app = demo.app

# --- Hot-reload: respawn ONLY the file-manager process, no Space rebuild ---
@app.get("/reload")
def reload_fm():
    start_fm()
    time.sleep(1)
    return HTMLResponse('Reloaded file-manager process. <a href="/fm/">back to /fm</a>')

@app.get("/gpuinfo")
def gpuinfo():
    return PlainTextResponse(gpu_report())

@app.get("/llm")
def llm(q: str = "Say hello and tell me which GPU you are running on."):
    return PlainTextResponse(run_llm(q))

# --- Proxy: /fm/... -> 127.0.0.1:9001/fm/... (method + body forwarded) ---
@app.api_route("/fm", methods=["GET", "POST"])
@app.api_route("/fm/{path:path}", methods=["GET", "POST"])
async def proxy(request: Request, path: str = ""):
    url = f"http://127.0.0.1:{PORT}/fm/{path}"
    if request.url.query:
        url += "?" + request.url.query
    body = await request.body()
    fwd = {k: v for k, v in request.headers.items() if k.lower() != "host"}
    req = urllib.request.Request(url, data=body or None, method=request.method, headers=fwd)
    try:
        with urllib.request.urlopen(req) as r:
            data, status, headers = r.read(), r.status, r.headers
    except urllib.error.HTTPError as e:
        data, status, headers = e.read(), e.code, e.headers
    except urllib.error.URLError:
        return Response(content=b"file-manager not up yet on 9001", status_code=502)
    out = {k: v for k, v in headers.items()
           if k.lower() in ("set-cookie", "location", "content-type")}
    return Response(content=data, status_code=status, headers=out)

while True:
    time.sleep(3600)