import gradio as gr from huggingface_hub import hf_hub_download from llama_cpp import Llama print("Downloading model...") model_path = hf_hub_download( repo_id="svjay/sdpo-llama-3-8b", filename="llama-3-8b-instruct.Q4_K_M.gguf" ) print("Loading model...") llm = Llama(model_path=model_path, n_ctx=2048, n_threads=4, verbose=False) print("Model ready!") def chat(message, history): prompt = f"<|begin_of_text|><|start_header_id|>user<|end_header_id|>\n{message}<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n" response = llm(prompt, max_tokens=512, stop=["<|eot_id|>"]) return response["choices"][0]["text"].strip() gr.ChatInterface( chat, title="SDPO LLaMA 3 8B", description="Fine-tuned with SFT + SDPO self-distillation" ).launch()