bwdev1's picture
Update app.py
32a82ba verified
Raw
History Blame Contribute Delete
1.95 kB
import gradio as gr
from llama_cpp import Llama
# -----------------------------
# Load GGUF model
# -----------------------------
def load_gguf():
model_name = "TheBloke/Wizard-Vicuna-30B-Uncensored-GGUF"
gguf_file = "wizard-vicuna-30b-uncensored.Q4_K_M.gguf"
llm = Llama.from_pretrained(
repo_id=model_name,
filename=gguf_file,
n_ctx=2048,
n_threads=8,
n_gpu_layers=0, # CPU only
verbose=False,
)
return llm
# -----------------------------
# Chat function
# -----------------------------
def chat_gguf(llm, history, message):
# Build conversation prompt
prompt = ""
for user, bot in history:
prompt += f"User: {user}\nAssistant: {bot}\n"
prompt += f"User: {message}\nAssistant:"
# Run inference
result = llm(
prompt,
max_tokens=512,
temperature=0.7,
top_p=0.9,
)
reply = result["choices"][0]["text"].strip()
return reply
# -----------------------------
# Gradio UI
# -----------------------------
def create_interface():
with gr.Blocks(title="Wizard Vicuna 30B GGUF Chat") as demo:
gr.Markdown(
"""
# 🧙‍♂️ Wizard‑Vicuna‑30B (GGUF)
Running on CPU using llama.cpp
"""
)
chatbot = gr.Chatbot(height=500)
msg = gr.Textbox(label="Your message")
clear = gr.Button("Clear Chat")
# Load model once at startup
state_model = gr.State(load_gguf())
# Chat handler
def respond(message, history, model):
reply = chat_gguf(model, history, message)
history.append([message, reply])
return history
msg.submit(
respond,
inputs=[msg, chatbot, state_model],
outputs=chatbot,
)
clear.click(lambda: [], None, chatbot)
return demo
demo = create_interface()
demo.launch()