File size: 2,459 Bytes
e3d36c9
d62d363
e3d36c9
e71ecba
 
 
 
 
 
 
 
 
 
 
 
 
6bd447e
e71ecba
d62d363
 
 
 
25612ec
d62d363
 
 
 
e3d36c9
 
 
 
 
 
 
 
 
 
 
d62d363
e3d36c9
 
 
 
25612ec
d62d363
 
 
 
 
 
 
 
 
 
 
 
 
e3d36c9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
25612ec
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
import gradio as gr
from huggingface_hub import hf_hub_download

import subprocess
import sys, platform
from importlib import metadata as md


#Install and Compile wheel at cost of 5minutes
subprocess.run("pip install -V llama_cpp_python==0.3.15", shell=True)

#Add Log to show all versions 
print("Python:", platform.python_version(), sys.implementation.name)
print("OS:", platform.uname())
print("\n".join(sorted(f"{d.metadata['Name']}=={d.version}" for d in md.distributions())))

from llama_cpp import Llama

# Download the GGUF model file from the repo
model_repo = "Molchevsky/ai_resume"
model_filename = "merged-Q6_K.gguf"  
model_path = hf_hub_download(repo_id=model_repo, filename=model_filename)

# Load the model once (outside the function for efficiency)
# Use chat_format="llama-3" since it's based on Llama 3.2
# Adjust n_ctx if needed for context length
llm = Llama(model_path, chat_format="llama-3", n_ctx=2048)

def respond(
    message,
    history: list[dict[str, str]],
    system_message,
    max_tokens,
    temperature,
    top_p,
):
    messages = [{"role": "system", "content": system_message}]

    # Extend with history (which is list of dicts with 'role' and 'content')
    messages.extend(history)

    messages.append({"role": "user", "content": message})

    response = ""

    # Use create_chat_completion with stream=True
    for chunk in llm.create_chat_completion(
        messages,
        max_tokens=max_tokens,
        temperature=temperature,
        top_p=top_p,
        stream=True,
    ):
        if 'content' in chunk['choices'][0]['delta']:
            token = chunk['choices'][0]['delta']['content']
            response += token
            yield response

"""
For information on how to customize the ChatInterface, peruse the gradio docs: https://www.gradio.app/docs/chatinterface
"""
chatbot = gr.ChatInterface(
    respond,
    type="messages",
    additional_inputs=[
        gr.Textbox(value="You are a friendly Chatbot.", label="System message"),
        gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="Max new tokens"),
        gr.Slider(minimum=0.1, maximum=4.0, value=0.7, step=0.1, label="Temperature"),
        gr.Slider(
            minimum=0.1,
            maximum=1.0,
            value=0.95,
            step=0.05,
            label="Top-p (nucleus sampling)",
        ),
    ],
)

with gr.Blocks() as demo:
    chatbot.render()

if __name__ == "__main__":
    demo.launch(debug=True)