Valtry commited on
Commit
5e9bfc3
·
verified ·
1 Parent(s): 1b56c12

Create app.py

Browse files
Files changed (1) hide show
  1. app.py +65 -0
app.py ADDED
@@ -0,0 +1,65 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ from fastapi import FastAPI
3
+ from fastapi.responses import StreamingResponse
4
+ from pydantic import BaseModel
5
+
6
+ from llama_cpp import Llama
7
+ from huggingface_hub import hf_hub_download
8
+
9
+ # 🔥 CONFIG
10
+ REPO_ID = "Valtry/Gemma-4" # change this
11
+ FILENAME = "google_gemma-4-E2B-it-Q4_K_M.gguf"
12
+
13
+ # 📥 Download model from HF
14
+ model_path = hf_hub_download(
15
+ repo_id=REPO_ID,
16
+ filename=FILENAME
17
+ )
18
+
19
+ # ⚡ Load model
20
+ llm = Llama(
21
+ model_path=model_path,
22
+ n_ctx=2048,
23
+ n_threads=4, # adjust based on CPU
24
+ n_gpu_layers=0 # CPU only (HF free tier)
25
+ )
26
+
27
+ # -------- FastAPI --------
28
+ app = FastAPI()
29
+
30
+ class Request(BaseModel):
31
+ prompt: str
32
+
33
+ # -------- Streaming generator --------
34
+ def stream_generate(prompt):
35
+ formatted_prompt = f"<start_of_turn>user\n{prompt}\n<end_of_turn>\n<start_of_turn>model\n"
36
+
37
+ output = llm(
38
+ formatted_prompt,
39
+ max_tokens=256,
40
+ temperature=0.7,
41
+ top_p=0.9,
42
+ stream=True
43
+ )
44
+
45
+ for chunk in output:
46
+ if "choices" in chunk:
47
+ token = chunk["choices"][0]["text"]
48
+ yield token
49
+
50
+ # -------- API endpoint --------
51
+ @app.post("/generate")
52
+ def generate(req: Request):
53
+ return StreamingResponse(stream_generate(req.prompt), media_type="text/plain")
54
+
55
+ # -------- Gradio UI --------
56
+ def chat_fn(message, history):
57
+ response = ""
58
+ for token in stream_generate(message):
59
+ response += token
60
+ yield response
61
+
62
+ ui = gr.ChatInterface(chat_fn)
63
+
64
+ # Mount UI
65
+ app = gr.mount_gradio_app(app, ui, path="/")