import torch from transformers import AutoTokenizer, AutoModelForCausalLM from peft import PeftModel import streamlit as st # Load base model and tokenizer base_model = "TinyLlama/TinyLlama-1.1B-Chat-v1.0" tokenizer = AutoTokenizer.from_pretrained(base_model) # Load base model on CPU model = AutoModelForCausalLM.from_pretrained( base_model, torch_dtype=torch.float32, device_map="cpu" ) # Load LoRA adapter model = PeftModel.from_pretrained(model, "lora_adapter", device_map="cpu") model.eval() # Prompt template def format_prompt(instruction): return f"""### SYSTEM: You are a helpful and expert Python programming tutor. You only answer questions related to Python programming. If the question is unrelated to Python, say: "Sorry, I can only answer Python-related questions." ### USER: {instruction} ### ASSISTANT: """ # Chat function def chat(instruction): prompt = format_prompt(instruction) inputs = tokenizer(prompt, return_tensors="pt").to(model.device) with torch.no_grad(): outputs = model.generate( **inputs, max_new_tokens=512, do_sample=False, temperature=0.7, top_p=0.9, repetition_penalty=1.1 ) response = tokenizer.decode(outputs[0], skip_special_tokens=True) # Debugging optional: # print("🧠 Full output:", repr(response)) # Split and return the assistant's answer if "### Answer:" in response: return response.split("### Answer:")[-1].strip() else: return response.strip() # Streamlit UI st.set_page_config(page_title="🐍 Python Tutor Chatbot") st.title("🐍 Python Tutor Chatbot(LoRA") st.write("Ask me Python programming questions!") user_input = st.text_area("Your question:") if st.button("Get Answer") and user_input.strip(): with st.spinner("Thinking..."): response = chat(user_input) st.markdown("**Answer:**") st.write(response)