llama_cpp / app.py
srinigith's picture
Llama.cpp changes
0e98c39
Raw
History Blame Contribute Delete
683 Bytes
from llama_cpp import Llama
import gradio as gr
from huggingface_hub import hf_hub_download
MODEL_REPO = "lmstudio-community/gemma-2-2b-it-GGUF"
MODEL_FILE = "gemma-2-2b-it-Q4_K_M.gguf"
model_path = hf_hub_download(
repo_id=MODEL_REPO,
filename=MODEL_FILE
)
llm = Llama(
model_path=model_path,
n_ctx=2048,
n_threads=4,
verbose=False
)
def chat(message, history):
prompt = f"User: {message}\nAssistant:"
output = llm(
prompt,
max_tokens=256,
stop=["User:", "\n\n"],
echo=False
)
return output["choices"][0]["text"].strip()
demo = gr.ChatInterface(
fn=chat,
title="Gemma 2 Chatbot"
)
demo.launch()