Operon / app.py
soumo-hug's picture
Update app.py
2ffc227 verified
Raw
History Blame Contribute Delete
2.08 kB
import os
import subprocess
import time
import requests
import gradio as gr
from huggingface_hub import hf_hub_download
MODEL_REPO = "unsloth/Qwen3.5-9B-GGUF"
MODEL_FILE = "Qwen3.5-9B-Q4_K_M.gguf"
LLAMA_DIR = "llama.cpp"
SERVER_PORT = "8000"
def install_dependencies():
subprocess.run(["apt-get", "update"])
subprocess.run([
"apt-get", "install", "-y",
"build-essential",
"cmake",
"git"
])
def download_model():
return hf_hub_download(
repo_id=MODEL_REPO,
filename=MODEL_FILE,
repo_type="model"
)
def setup_llama():
if not os.path.exists(LLAMA_DIR):
subprocess.run([
"git", "clone",
"https://github.com/ggml-org/llama.cpp"
], check=True)
os.makedirs(f"{LLAMA_DIR}/build", exist_ok=True)
subprocess.run([
"cmake",
"-B", "build"
], cwd=LLAMA_DIR, check=True)
subprocess.run([
"cmake",
"--build",
"build",
"-j"
], cwd=LLAMA_DIR, check=True)
def start_server(model_path):
subprocess.Popen([
"./build/bin/llama-server",
"-m", model_path,
"--port", SERVER_PORT,
"-c", "4096",
"-t", "8"
], cwd=LLAMA_DIR)
def wait_for_server():
url = f"http://localhost:{SERVER_PORT}/health"
for _ in range(60):
try:
requests.get(url, timeout=1)
return
except:
time.sleep(2)
def chat(prompt):
r = requests.post(
f"http://localhost:{SERVER_PORT}/v1/chat/completions",
json={
"model": "qwen3.5-9b",
"messages": [
{"role": "user", "content": prompt}
]
},
timeout=300
)
data = r.json()
return data["choices"][0]["message"]["content"]
install_dependencies()
model_path = download_model()
setup_llama()
start_server(model_path)
wait_for_server()
ui = gr.Interface(
fn=chat,
inputs=gr.Textbox(lines=5),
outputs="text",
title="Operon Dev LLM (Qwen3.5-9B)"
)
ui.launch(server_name="0.0.0.0", server_port=7860)