Spaces:
Sleeping
Sleeping
Download app.py from yk07577/Assignment2: direct link, hf CLI and curl.
- Browser
- Download file 3.26 kB
-
https://huggingface.co/spaces/yk07577/Assignment2/resolve/main/app.py
- Command line
-
hf download hf://spaces/yk07577/Assignment2/app.py
-
curl -L -o app.py https://huggingface.co/spaces/yk07577/Assignment2/resolve/main/app.py
3.26 kB
| import gradio as gr | |
| import torch | |
| from transformers import AutoTokenizer, AutoModelForCausalLM | |
| import math | |
| # Default public model | |
| DEFAULT_MODEL = "HuggingFaceH4/zephyr-7b-beta" | |
| def run_analysis(model_id, prompt_variations, temperature, max_new_tokens): | |
| # Load model | |
| tokenizer = AutoTokenizer.from_pretrained(model_id, use_fast=True) | |
| model = AutoModelForCausalLM.from_pretrained(model_id, torch_dtype=torch.float16 if torch.cuda.is_available() else torch.float32, device_map="auto") | |
| results = [] | |
| for prompt in prompt_variations: | |
| inputs = tokenizer(prompt, return_tensors="pt").to(model.device) | |
| with torch.no_grad(): | |
| outputs = model.generate( | |
| **inputs, | |
| max_new_tokens=max_new_tokens, | |
| temperature=temperature, | |
| do_sample=True if temperature > 0 else False, | |
| return_dict_in_generate=True, | |
| output_scores=True | |
| ) | |
| # Decode text | |
| output_text = tokenizer.decode(outputs.sequences[0], skip_special_tokens=True) | |
| # Extract generated part | |
| prompt_len = inputs["input_ids"].shape[1] | |
| gen_ids = outputs.sequences[0][prompt_len:] | |
| scores = outputs.scores | |
| if len(scores) != len(gen_ids): | |
| L = min(len(scores), len(gen_ids)) | |
| scores = scores[:L] | |
| gen_ids = gen_ids[:L] | |
| # Logprobs | |
| logprobs = [] | |
| for step_logits, tid in zip(scores, gen_ids): | |
| lp = torch.log_softmax(step_logits, dim=-1)[tid.item()].item() | |
| logprobs.append(lp) | |
| # Build table of top-5 alternatives | |
| table = "Token | P(token) | Top alternatives\n" | |
| table += "-"*50 + "\n" | |
| for tid, lp, step_logits in zip(gen_ids, logprobs, scores): | |
| alts = torch.topk(torch.log_softmax(step_logits, dim=-1), 5) | |
| alt_tokens = [tokenizer.decode([i]) for i in alts.indices.tolist()] | |
| alt_probs = [f"{math.exp(p)*100:.1f}%" for p in alts.values.tolist()] | |
| alt_str = ", ".join([f"{t} ({p})" for t, p in zip(alt_tokens, alt_probs)]) | |
| table += f"{tokenizer.decode([tid])} | {math.exp(lp)*100:.1f}% | {alt_str}\n" | |
| results.append(f"Prompt: {prompt}\n\nOutput:\n{output_text}\n\nToken Probabilities:\n{table}") | |
| return "\n\n---\n\n".join(results) | |
| # Gradio UI | |
| with gr.Blocks() as demo: | |
| gr.Markdown("# Prompt Variations & Token Analysis (Free, HF Models)") | |
| model_id = gr.Textbox(value=DEFAULT_MODEL, label="Model ID", placeholder="Enter a model like HuggingFaceH4/zephyr-7b-beta") | |
| prompts = gr.Textbox(lines=6, value="Explain dollar-cost averaging to a beginner in 6–8 sentences.\nIn 6–8 sentences, teach a newbie how dollar-cost averaging works.", label="Prompt Variations (one per line)") | |
| temp = gr.Slider(0, 1, 0.2, label="Temperature") | |
| max_tokens = gr.Slider(10, 500, 220, step=10, label="Max new tokens") | |
| btn = gr.Button("Run Analysis") | |
| output = gr.Textbox(lines=30, label="Results") | |
| btn.click( | |
| run_analysis, | |
| inputs=[model_id, prompts, temp, max_tokens], | |
| outputs=output | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch() | |