| import gradio as gr |
| from llama_cpp import Llama |
| from huggingface_hub import hf_hub_download |
|
|
| |
| llm = None |
|
|
| def load_model_if_needed(): |
| global llm |
| if llm is None: |
| |
| print("Scaricamento e caricamento del modello in corso (attendi)...") |
| model_path = hf_hub_download( |
| repo_id="FPll/limba-mentor-llama3-gguf", |
| filename="Meta-Llama-3.1-8B.Q4_K_M.gguf" |
| ) |
| llm = Llama( |
| model_path=model_path, |
| n_ctx=4096, |
| n_threads=2, |
| verbose=False |
| ) |
| return llm |
|
|
| def generate_response(message, history): |
| |
| model = load_model_if_needed() |
| |
| messages = [ |
| {"role": "system", "content": "Sei Limba, un assistente alla programmazione esperto, diretto e preciso."} |
| ] |
| |
| for user_msg, bot_msg in history: |
| messages.append({"role": "user", "content": user_msg}) |
| messages.append({"role": "assistant", "content": bot_msg}) |
| |
| messages.append({"role": "user", "content": message}) |
| |
| response = model.create_chat_completion( |
| messages=messages, |
| max_tokens=512, |
| stream=True |
| ) |
| |
| partial_message = "" |
| for chunk in response: |
| if "choices" in chunk and len(chunk["choices"]) > 0: |
| delta = chunk["choices"][0].get("delta", {}) |
| if "content" in delta: |
| partial_message += delta["content"] |
| yield partial_message |
|
|
| demo = gr.ChatInterface( |
| fn=generate_response, |
| title="Limba 2.0", |
| description="Il tuo assistente esperto. Nota: Il PRIMO messaggio richiederà 1-2 minuti di attesa per avviare il motore interno.", |
| examples=["Come posso ottimizzare un loop in Python?"], |
| theme="soft" |
| ) |
|
|
| if __name__ == "__main__": |
| demo.launch() |