Spaces:
Running on Zero
Running on Zero
File size: 3,534 Bytes
63c78de 11fbc99 63c78de 4d7ebea 63c78de 11fbc99 4d7ebea 63c78de 11fbc99 63c78de 11fbc99 63c78de 97c5d23 63c78de 11fbc99 0bb9433 63c78de 0bb9433 63c78de 0bb9433 63c78de 97c5d23 63c78de 0bb9433 63c78de 7aac6c0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 | import spaces
import torch
import gradio as gr
from transformers import AutoModelForCausalLM, AutoTokenizer
from peft import PeftModel
SYSTEM = (
"You are the Nodrix build assistant. You help ESP32 and Arduino developers "
"build projects with the Nodrix library. Use only real Nodrix APIs."
)
BASE_7B = "Qwen/Qwen2.5-Coder-7B-Instruct"
BASE_15B = "Qwen/Qwen2.5-Coder-1.5B-Instruct"
_avail = torch.cuda.is_available
torch.cuda.is_available = lambda: False
tok7 = AutoTokenizer.from_pretrained(BASE_7B)
m7 = AutoModelForCausalLM.from_pretrained(BASE_7B, dtype=torch.bfloat16)
m7 = PeftModel.from_pretrained(m7, "decoded-cipher/nodrix-coder-7b-lora-v2", adapter_name="v2")
m7.load_adapter("decoded-cipher/nodrix-coder-7b-lora-v3", adapter_name="v3")
torch.cuda.is_available = _avail
m7 = m7.to("cuda")
_m15 = None
def base15():
global _m15
if _m15 is None:
tok = AutoTokenizer.from_pretrained(BASE_15B)
model = AutoModelForCausalLM.from_pretrained(BASE_15B, dtype=torch.bfloat16).to("cuda")
model = PeftModel.from_pretrained(model, "decoded-cipher/nodrix-coder-1.5b-lora-v1", adapter_name="v1")
_m15 = (model, tok)
return _m15
MODELS = {
"Base 7B (no fine-tune)": ("7b", None),
"v2 · 7B": ("7b", "v2"),
"v3 · 7B ★": ("7b", "v3"),
"Base 1.5B (no fine-tune)": ("15b", None),
"v1 · 1.5B": ("15b", "v1"),
}
CHOICES = list(MODELS)
@spaces.GPU(duration=45)
def run(choice, question):
which, adapter = MODELS[choice]
model, tok = (m7, tok7) if which == "7b" else base15()
inputs = tok.apply_chat_template(
[{"role": "system", "content": SYSTEM}, {"role": "user", "content": question}],
add_generation_prompt=True, return_tensors="pt", return_dict=True,
).to("cuda")
n = inputs["input_ids"].shape[1]
def gen():
out = model.generate(**inputs, max_new_tokens=300, do_sample=True,
temperature=0.7, top_p=0.9, pad_token_id=tok.eos_token_id)
return tok.decode(out[0][n:], skip_special_tokens=True)
if adapter is None:
with model.disable_adapter():
return gen()
model.set_adapter(adapter)
return gen()
def compare(question, left, right):
return run(left, question), run(right, question)
EXAMPLES = [
"How do I control a relay from the dashboard?",
"Write a Nodrix sketch: dim an LED from a slider widget bound to \"brightness\".",
"How do I build a battery sensor that sleeps between readings?",
"Is nodrix suitable for commercial or industrial projects?",
]
with gr.Blocks(title="Nodrix build assistant — base vs fine-tuned") as demo:
gr.Markdown(
"# Nodrix build assistant — base vs fine-tuned\n"
"Three LoRA fine-tunes of Qwen2.5-Coder for the Nodrix ESP32/Arduino library, "
"side by side with the base model. Pick a model per column and ask the same "
"question. `★` v3 is the best run. Free ZeroGPU — the first call cold-starts."
)
question = gr.Textbox(label="Question", lines=2, value=EXAMPLES[0])
with gr.Row():
left = gr.Dropdown(CHOICES, value="Base 7B (no fine-tune)", label="Left model")
right = gr.Dropdown(CHOICES, value="v3 · 7B ★", label="Right model")
go = gr.Button("Compare", variant="primary")
with gr.Row():
out_l = gr.Markdown()
out_r = gr.Markdown()
gr.Examples(EXAMPLES, inputs=question)
go.click(compare, [question, left, right], [out_l, out_r])
demo.queue().launch(ssr_mode=False)
|