import os import gradio as gr from transformers import pipeline print("\n#### Example: Deploying a Simple LLM with Gradio to Hugging Face Spaces") print("The following code demonstrates a simple LLM inference application using Gradio, which can be easily deployed to Hugging Face Spaces. For this example, we'll use a small, efficient model.") # --- Step 1: Load the LLM --- # # We'll use a small text generation model for demonstration purposes. # For larger models, ensure your Hugging Face Space has sufficient GPU resources. print("\nLoading text generation pipeline...") # You might need to install 'accelerate' and 'bitsandbytes' for quantized models or faster inference # !pip install -q accelerate bitsandbytes # This model is small and runs well on CPU for basic demonstrations. # For a more capable LLM, you'd choose models like 'google/gemma-2b-it' or 'TinyLlama/TinyLlama-1.1B-Chat-v1.0' # but these would require a GPU runtime on your Space. # For this example, let's pick a very small model for fast execution even on CPU. # For a true LLM, consider a model like 'distilbert-base-uncased' for text classification or 'gpt2' for generation, # though 'gpt2' can be slow without GPU. # Let's use a simple sentiment analysis model as it's quick and illustrative. # If a full text generation LLM is desired, 'gpt2' is a good choice if a GPU is available or patience is high. # For a quick CPU example of a small text model, let's use a fill-mask pipeline. # Fallback for LLM-like behavior with quick execution (e.g., fill-mask) # A true 'text-generation' LLM would be 'gpt2' but it's slower. # Using a small 'fill-mask' model to simulate interactive LLM-like behavior quickly. # If you have GPU, you can switch to: pipeline("text-generation", model="gpt2") try: # Attempt to load a text-generation model first generator = pipeline("text-generation", model="distilgpt2", device=0) # device=0 for GPU if available print("Using distilgpt2 for text generation.") except Exception as e: print(f"Could not load distilgpt2 for text-generation, falling back to fill-mask: {e}") generator = pipeline("fill-mask", model="distilbert-base-uncased") print("Using distilbert-base-uncased for fill-mask.") print("Model loaded.") # --- Step 2: Define the LLM inference function --- # def llm_inference(prompt): if generator.task == "text-generation": response = generator(prompt, max_new_tokens=50, num_return_sequences=1) return response[0]['generated_text'] elif generator.task == "fill-mask": response = generator(prompt) # For fill-mask, return the top predicted token and its score top_prediction = response[0] return f"{prompt.replace('[MASK]', f'**{top_prediction['token_str']}**')} (Score: {top_prediction['score']:.2f})" # --- Step 3: Create a Gradio Interface --- # print("\nBuilding Gradio interface...") if generator.task == "text-generation": interface = gr.Interface( fn=llm_inference, inputs=gr.Textbox(lines=2, placeholder="Enter your prompt here..."), outputs=gr.Textbox(lines=5), title="Simple LLM Chatbot (distilgpt2)", description="Enter a prompt and get text generated by a small LLM." ) else: interface = gr.Interface( fn=llm_inference, inputs=gr.Textbox(lines=2, placeholder="Enter a sentence with [MASK] to fill, e.g., 'The capital of France is [MASK].'"), outputs=gr.Textbox(lines=5), title="Masked Language Model (distilbert)", description="Enter a sentence with a [MASK] token to see the model's prediction." ) # --- Step 4: Launch the Gradio App --- # import gradio as gr from transformers import pipeline import os from huggingface_hub import login # Authenticate with HF Hub hf_token = os.getenv("HF_TOKEN") if hf_token: login(token=hf_token) # Load a small, efficient model generator = pipeline("text-generation", model="distilgpt2", token=hf_token) def generate_text(prompt): result = generator(prompt, max_length=50, num_return_sequences=1) return result[0]["generated_text"] demo = gr.Interface( fn=generate_text, inputs="text", outputs="text", title="Simple LLM Demo", description="A lightweight text generation app using DistilGPT2." ) # Keep the app alive on Hugging Face Spaces demo.launch(server_name="0.0.0.0", server_port=7860)