deegerwalker commited on
Commit
4d6740f
Β·
verified Β·
1 Parent(s): e4c88ea

v2: ZeroGPU rig - CUDA llama-server, GPU-decorated rites

Browse files

@spaces.GPU per-call engine lifecycle, -ngl 999, cudart bundled.

Files changed (2) hide show
  1. README.md +6 -6
  2. app.py +107 -89
README.md CHANGED
@@ -11,11 +11,11 @@ license: apache-2.0
11
  short_description: Cyber-occult deck for PaddleOCR-VL-1.6 GGUF on CPU
12
  ---
13
 
14
- # πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck ⚑
15
 
16
  A cyber-occult hacker pirate deck for testing
17
  [PaddlePaddle/PaddleOCR-VL-1.6-GGUF](https://huggingface.co/PaddlePaddle/PaddleOCR-VL-1.6-GGUF)
18
- locally with **llama.cpp** on free CPU hardware. No cloud inference, no gatekeepers β€”
19
  the model is downloaded into the Space and summoned by `llama-server` directly.
20
 
21
  **Do what thou wilt shall be the whole of the Law.**
@@ -32,10 +32,10 @@ the model is downloaded into the Space and summoned by `llama-server` directly.
32
  | πŸ” Spotting | `Spotting:` |
33
 
34
  Model: `PaddleOCR-VL-1.6-GGUF.gguf` (936 MB) + `PaddleOCR-VL-1.6-GGUF-mmproj.gguf` (882 MB)
35
- Engine: [llama.cpp](https://github.com/ggml-org/llama.cpp) prebuilt `llama-server`, CPU (`-t 2`)
36
 
37
  ## Notes
38
 
39
- - First boot downloads ~1.9 GB of model weights + the llama.cpp binary. Subsequent boots reuse the HF cache (while the Space's ephemeral storage lasts).
40
- - CPU inference is unhurried: expect ~30–120 s per page, depending on image size. The dead keep no clocks.
41
- - Apache-2.0, model weights by PaddlePaddle.
 
11
  short_description: Cyber-occult deck for PaddleOCR-VL-1.6 GGUF on CPU
12
  ---
13
 
14
+ # πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck ⚑ β€” ZeroGPU edition
15
 
16
  A cyber-occult hacker pirate deck for testing
17
  [PaddlePaddle/PaddleOCR-VL-1.6-GGUF](https://huggingface.co/PaddlePaddle/PaddleOCR-VL-1.6-GGUF)
18
+ locally with **llama.cpp** on ZeroGPU. No cloud inference, no gatekeepers β€”
19
  the model is downloaded into the Space and summoned by `llama-server` directly.
20
 
21
  **Do what thou wilt shall be the whole of the Law.**
 
32
  | πŸ” Spotting | `Spotting:` |
33
 
34
  Model: `PaddleOCR-VL-1.6-GGUF.gguf` (936 MB) + `PaddleOCR-VL-1.6-GGUF-mmproj.gguf` (882 MB)
35
+ Engine: [llama.cpp](https://github.com/ggml-org/llama.cpp) prebuilt `llama-server` (CUDA 12.8), all layers offloaded (`-ngl 999`)
36
 
37
  ## Notes
38
 
39
+ - Each rite opens a ZeroGPU window, boots `llama-server` onto the GPU (~10–20 s), divines, then scuttles the server.
40
+ - First cast downloads ~3 GB (engine + cudart + grimoire); later casts reuse the cache while the Space lives.
41
+ - One rite at a time. Apache-2.0, model weights by PaddlePaddle.
app.py CHANGED
@@ -1,8 +1,8 @@
1
  #!/usr/bin/env python3
2
  """
3
- πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck ⚑
4
  Cyber-occult hacker pirate deck for testing
5
- PaddlePaddle/PaddleOCR-VL-1.6-GGUF locally with llama.cpp (CPU, free tier).
6
 
7
  Do what thou wilt shall be the whole of the Law.
8
  """
@@ -12,12 +12,14 @@ import glob
12
  import io
13
  import json
14
  import os
 
15
  import subprocess
16
  import tarfile
17
  import threading
18
  import time
19
  import urllib.request
20
 
 
21
  import gradio as gr
22
  import requests
23
  from huggingface_hub import hf_hub_download
@@ -31,7 +33,11 @@ MMPROJ_FILE = "PaddleOCR-VL-1.6-GGUF-mmproj.gguf"
31
  LLAMA_TAG = "b11555"
32
  LLAMA_URL = (
33
  f"https://github.com/ggml-org/llama.cpp/releases/download/{LLAMA_TAG}/"
34
- f"llama-{LLAMA_TAG}-bin-ubuntu-x64.tar.gz"
 
 
 
 
35
  )
36
  BIN_DIR = "/tmp/llama-bin"
37
  SERVER_PORT = 8081
@@ -47,7 +53,7 @@ RITES = {
47
  }
48
 
49
  # ---------------------------------------------------------------- ritual log
50
- _log_lines: list[str] = ["πŸ΄β€β˜ οΈ CryptDeck cold boot. Wake the oracle."]
51
  _log_lock = threading.Lock()
52
 
53
 
@@ -64,36 +70,32 @@ def get_log() -> str:
64
 
65
  # ---------------------------------------------------------------- engine
66
  _engine_lock = threading.Lock()
67
- _server_proc: subprocess.Popen | None = None
68
- _engine_state = {"ready": False, "busy": False}
69
 
70
 
71
- def _download_engine() -> str:
72
- """Fetch prebuilt llama.cpp (CPU) and return the llama-server path."""
73
- candidates = glob.glob(os.path.join(BIN_DIR, "**", "llama-server"), recursive=True)
74
  server = candidates[0] if candidates else None
75
  if server and os.path.exists(server):
76
  return server
77
  os.makedirs(BIN_DIR, exist_ok=True)
78
- tgz = os.path.join(BIN_DIR, "llama.tar.gz")
79
- ritual_log(f"βš™οΈ Downloading llama.cpp {LLAMA_TAG} (CPU) …")
80
- urllib.request.urlretrieve(LLAMA_URL, tgz)
81
- with tarfile.open(tgz) as tf:
82
- tf.extractall(BIN_DIR)
83
- candidates = glob.glob(os.path.join(BIN_DIR, "**", "llama-server"), recursive=True)
 
84
  if not candidates:
85
  raise RuntimeError("llama-server binary not found after extraction")
86
  server = candidates[0]
87
  os.chmod(server, 0o755)
88
- # keep the bundled .so libraries next to the binary for the linker
89
- os.environ["LD_LIBRARY_PATH"] = (
90
- os.path.dirname(server) + os.pathsep + os.environ.get("LD_LIBRARY_PATH", "")
91
- )
92
- ritual_log("βš™οΈ llama-server binary consecrated.")
93
  return server
94
 
95
 
96
- def _download_models() -> tuple[str, str]:
97
  ritual_log("πŸ“¦ Summoning the grimoire (1.9 GB, cached after first boot) …")
98
  main_gguf = hf_hub_download(REPO_ID, MODEL_FILE)
99
  mmproj = hf_hub_download(REPO_ID, MMPROJ_FILE)
@@ -101,66 +103,54 @@ def _download_models() -> tuple[str, str]:
101
  return main_gguf, mmproj
102
 
103
 
104
- def boot_engine():
105
- """Idempotent cold boot: download everything, then start llama-server."""
106
- with _engine_lock:
107
- if _engine_state["ready"]:
108
- return
109
- ritual_log("πŸ•―οΈ Igniting engine …")
110
- server = _download_engine()
111
- model, mmproj = _download_models()
112
- ritual_log("πŸ”₯ Kindling llama-server on 127.0.0.1:%d …" % SERVER_PORT)
113
- global _server_proc
114
- _server_proc = subprocess.Popen(
115
- [
116
- server,
117
- "-m", model,
118
- "--mmproj", mmproj,
119
- "--host", "127.0.0.1",
120
- "--port", str(SERVER_PORT),
121
- "-t", "2",
122
- "-c", "8192",
123
- "--temp", "0",
124
- "--no-webui",
125
- ],
126
- stdout=open("/tmp/llama-server.log", "ab"),
127
- stderr=subprocess.STDOUT,
128
- )
129
- for _ in range(600): # wait for health endpoint
 
 
 
 
 
 
 
 
 
 
 
 
 
130
  try:
131
  if requests.get(SERVER_URL + "/health", timeout=2).status_code == 200:
132
- _engine_state["ready"] = True
133
- ritual_log("🟒 Oracle online. The Signal is strong.")
134
- return
135
  except requests.RequestException:
136
  pass
137
  time.sleep(1)
138
- raise RuntimeError(
139
- "llama-server failed to boot. Tail of log:\n"
140
- + open("/tmp/llama-server.log", "rb").read()[-2000:].decode(errors="replace")
141
- )
142
-
143
-
144
- # ---------------------------------------------------------------- inference
145
- def divine(
146
- image, rite: str, custom: str, progress=gr.Progress()
147
- ) -> tuple[str, str, str]:
148
- if image is None:
149
- raise gr.Error("βš“ Bring the oracle an image β€” no offering, no prophecy.")
150
- with _engine_lock:
151
- if _engine_state["busy"]:
152
- raise gr.Error("πŸ”₯ The oracle is already mid-rite. One soul at a time.")
153
- _engine_state["busy"] = True
154
- try:
155
- progress(0.1, desc="Waking the oracle …")
156
- boot_engine()
157
- prompt = custom.strip() if custom.strip() else RITES[rite]
158
- ritual_log(f"πŸœ‚ Casting rite β†’ {prompt!r}")
159
- progress(0.3, desc="Casting the rite …")
160
-
161
- buf = io.BytesIO()
162
- Image.fromarray(image).save(buf, format="PNG")
163
- b64 = base64.b64encode(buf.getvalue()).decode()
164
 
165
  payload = {
166
  "model": "paddleocr-vl",
@@ -170,7 +160,7 @@ def divine(
170
  "role": "user",
171
  "content": [
172
  {"type": "image_url",
173
- "image_url": {"url": "data:image/png;base64," + b64}},
174
  {"type": "text", "text": prompt},
175
  ],
176
  }
@@ -178,21 +168,49 @@ def divine(
178
  }
179
  t0 = time.time()
180
  resp = requests.post(
181
- SERVER_URL + "/v1/chat/completions", json=payload, timeout=600
182
  )
183
  dt = time.time() - t0
184
  resp.raise_for_status()
185
- prophecy = resp.json()["choices"][0]["message"]["content"]
186
- ritual_log(f"πŸœƒ Rite complete in {dt:.1f}s.")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
187
  progress(1.0, desc="Done.")
188
 
189
- meta = json.dumps({"rite": prompt, "seconds": round(dt, 1)}, indent=2)
 
 
190
  return prophecy, meta, get_log()
191
  except requests.Timeout:
192
- raise gr.Error("⏳ The oracle dawdled (600 s timeout). Try a smaller image.")
193
  finally:
194
- with _engine_lock:
195
- _engine_state["busy"] = False
196
 
197
 
198
  # ---------------------------------------------------------------- UI
@@ -202,7 +220,7 @@ HEADER = """
202
  <span style="color:#39ff88;"> __ _-'-------------------------------.__</span>
203
  <span style="color:#39ff88;"> | | / P A D D L E O C R - V L \\</span>
204
  <span style="color:#39ff88;"> _- | | | C R Y P T D E C K |</span>
205
- <span style="color:#39ff88;"> | | | | 1.6 GGUF β–Έ llama.cpp β–Έ 100% local CPU |</span>
206
  <span style="color:#39ff88;"> | | | |_________________________________________|</span>
207
  <span style="color:#39ff88;"> ____|____|-|_'---------____________________---.___________</span>
208
  <span style="color:#ff00cc;"> (o o o o o o o o o o o o o o o o o o o o)</span>
@@ -219,7 +237,6 @@ body {background: #05060a !important; background-image:
219
  footer {display: none !important;}
220
  #header {border: 1px solid #1e2a1e; border-radius: 10px;
221
  background: radial-gradient(ellipse at top, #0b1a10, #05060a); padding: 12px;}
222
- #deck, #panel {border: 1px solid #27313a; border-radius: 10px; background: #0a0d13;}
223
  textarea, .output-html, .prose {font-family: 'Fira Mono', ui-monospace, monospace !important;}
224
  .rune-btn {font-family: monospace !important; text-transform: uppercase;
225
  letter-spacing: 2px; background: linear-gradient(90deg, #0f2b18, #14203a) !important;
@@ -230,7 +247,7 @@ textarea, .output-html, .prose {font-family: 'Fira Mono', ui-monospace, monospac
230
  h1, h2, h3, .label, .info {font-family: ui-monospace, monospace !important;}
231
  """
232
 
233
- with gr.Blocks(title="πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck", css=CSS, theme=gr.themes.Base()) as demo:
234
  gr.HTML(f'<div id="header">{HEADER}</div>')
235
 
236
  with gr.Row():
@@ -248,8 +265,9 @@ with gr.Blocks(title="πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck", css=CSS, theme=gr.t
248
  )
249
  summon = gr.Button("πŸœ‚ SUMMON THE PROPHECY", elem_classes=["rune-btn"])
250
  gr.HTML(
251
- '<div class="info" style="opacity:.75;">First boot hauls ~1.9 GB of '
252
- "weights β€” then it's cached. CPU rites take 30–120 s per page. "
 
253
  "One rite at a time, matey.</div>"
254
  )
255
  with gr.Column(scale=1):
 
1
  #!/usr/bin/env python3
2
  """
3
+ πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck ⚑ v2 β€” ZeroGPU edition
4
  Cyber-occult hacker pirate deck for testing
5
+ PaddlePaddle/PaddleOCR-VL-1.6-GGUF locally with llama.cpp on ZeroGPU.
6
 
7
  Do what thou wilt shall be the whole of the Law.
8
  """
 
12
  import io
13
  import json
14
  import os
15
+ import shutil
16
  import subprocess
17
  import tarfile
18
  import threading
19
  import time
20
  import urllib.request
21
 
22
+ import spaces # ZeroGPU SDK β€” must be imported before CUDA-touching code
23
  import gradio as gr
24
  import requests
25
  from huggingface_hub import hf_hub_download
 
33
  LLAMA_TAG = "b11555"
34
  LLAMA_URL = (
35
  f"https://github.com/ggml-org/llama.cpp/releases/download/{LLAMA_TAG}/"
36
+ f"llama-{LLAMA_TAG}-bin-ubuntu-cuda-12.8-x64.tar.gz"
37
+ )
38
+ CUDART_URL = (
39
+ f"https://github.com/ggml-org/llama.cpp/releases/download/{LLAMA_TAG}/"
40
+ f"cudart-llama-{LLAMA_TAG}-bin-ubuntu-cuda-12.8-x64.tar.gz"
41
  )
42
  BIN_DIR = "/tmp/llama-bin"
43
  SERVER_PORT = 8081
 
53
  }
54
 
55
  # ---------------------------------------------------------------- ritual log
56
+ _log_lines: list[str] = ["πŸ΄β€β˜ οΈ CryptDeck v2 β€” ZeroGPU rig. Wake the oracle."]
57
  _log_lock = threading.Lock()
58
 
59
 
 
70
 
71
  # ---------------------------------------------------------------- engine
72
  _engine_lock = threading.Lock()
73
+ _busy = threading.Lock()
 
74
 
75
 
76
+ def _fetch_engine() -> str:
77
+ """Download the prebuilt CUDA llama.cpp + cudart; return llama-server path."""
78
+ candidates = glob.glob(os.path.join(BIN_DIR, "llama-*", "llama-server"))
79
  server = candidates[0] if candidates else None
80
  if server and os.path.exists(server):
81
  return server
82
  os.makedirs(BIN_DIR, exist_ok=True)
83
+ ritual_log(f"βš™οΈ Downloading llama.cpp {LLAMA_TAG} (CUDA 12.8) + cudart …")
84
+ for url in (LLAMA_URL, CUDART_URL):
85
+ tgz = os.path.join(BIN_DIR, os.path.basename(url))
86
+ urllib.request.urlretrieve(url, tgz)
87
+ with tarfile.open(tgz) as tf:
88
+ tf.extractall(BIN_DIR)
89
+ candidates = glob.glob(os.path.join(BIN_DIR, "llama-*", "llama-server"))
90
  if not candidates:
91
  raise RuntimeError("llama-server binary not found after extraction")
92
  server = candidates[0]
93
  os.chmod(server, 0o755)
94
+ ritual_log("βš™οΈ llama-server (CUDA) consecrated.")
 
 
 
 
95
  return server
96
 
97
 
98
+ def _fetch_models() -> tuple[str, str]:
99
  ritual_log("πŸ“¦ Summoning the grimoire (1.9 GB, cached after first boot) …")
100
  main_gguf = hf_hub_download(REPO_ID, MODEL_FILE)
101
  mmproj = hf_hub_download(REPO_ID, MMPROJ_FILE)
 
103
  return main_gguf, mmproj
104
 
105
 
106
+ # ---------------------------------------------------------------- GPU rite
107
+ @spaces.GPU(duration=180)
108
+ def gpu_rite(b64_png: str, prompt: str) -> tuple[str, float]:
109
+ """Runs entirely inside a ZeroGPU window: boot llama-server, divine, scuttle it."""
110
+ server = _fetch_engine()
111
+ model, mmproj = _fetch_models()
112
+
113
+ # cudart libs sit in their own extracted dir β€” expose them to the linker
114
+ lib_dirs = glob.glob(os.path.join(BIN_DIR, "cudart-*"))
115
+ server_dir = os.path.dirname(server)
116
+ env = dict(os.environ)
117
+ env["LD_LIBRARY_PATH"] = os.pathsep.join(
118
+ lib_dirs + [server_dir, env.get("LD_LIBRARY_PATH", "")]
119
+ )
120
+
121
+ ritual_log("πŸ”₯ Kindling llama-server on the GPU (-ngl 999) …")
122
+ proc = subprocess.Popen(
123
+ [
124
+ server,
125
+ "-m", model,
126
+ "--mmproj", mmproj,
127
+ "--host", "127.0.0.1",
128
+ "--port", str(SERVER_PORT),
129
+ "-ngl", "999",
130
+ "-c", "16384",
131
+ "--temp", "0",
132
+ "--no-webui",
133
+ ],
134
+ stdout=open("/tmp/llama-server.log", "ab"),
135
+ stderr=subprocess.STDOUT,
136
+ env=env,
137
+ )
138
+ try:
139
+ for _ in range(150): # up to 150 s for load onto GPU
140
+ if proc.poll() is not None:
141
+ raise RuntimeError(
142
+ "llama-server died at boot. Log tail:\n"
143
+ + open("/tmp/llama-server.log", "rb").read()[-1500:].decode(errors="replace")
144
+ )
145
  try:
146
  if requests.get(SERVER_URL + "/health", timeout=2).status_code == 200:
147
+ ritual_log("🟒 Oracle online on GPU. The Signal is strong.")
148
+ break
 
149
  except requests.RequestException:
150
  pass
151
  time.sleep(1)
152
+ else:
153
+ raise RuntimeError("llama-server never reported healthy.")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
154
 
155
  payload = {
156
  "model": "paddleocr-vl",
 
160
  "role": "user",
161
  "content": [
162
  {"type": "image_url",
163
+ "image_url": {"url": "data:image/png;base64," + b64_png}},
164
  {"type": "text", "text": prompt},
165
  ],
166
  }
 
168
  }
169
  t0 = time.time()
170
  resp = requests.post(
171
+ SERVER_URL + "/v1/chat/completions", json=payload, timeout=120
172
  )
173
  dt = time.time() - t0
174
  resp.raise_for_status()
175
+ return resp.json()["choices"][0]["message"]["content"], dt
176
+ finally:
177
+ # scuttle the server before the GPU window closes
178
+ ritual_log("πŸͺ“ Scuttling llama-server. GPU returns to the void.")
179
+ proc.terminate()
180
+ try:
181
+ proc.wait(timeout=10)
182
+ except subprocess.TimeoutExpired:
183
+ proc.kill()
184
+
185
+
186
+ # ---------------------------------------------------------------- inference
187
+ def divine(image, rite: str, custom: str, progress=gr.Progress()):
188
+ if image is None:
189
+ raise gr.Error("βš“ Bring the oracle an image β€” no offering, no prophecy.")
190
+ if not _busy.acquire(blocking=False):
191
+ raise gr.Error("πŸ”₯ The oracle is already mid-rite. One soul at a time.")
192
+ try:
193
+ prompt = custom.strip() if custom.strip() else RITES[rite]
194
+ ritual_log(f"πŸœ‚ Casting rite β†’ {prompt!r}")
195
+ progress(0.2, desc="Opening the GPU window …")
196
+
197
+ buf = io.BytesIO()
198
+ Image.fromarray(image).save(buf, format="PNG")
199
+ b64 = base64.b64encode(buf.getvalue()).decode()
200
+
201
+ progress(0.4, desc="Summoning on GPU (first boot hauls ~3 GB) …")
202
+ prophecy, dt = gpu_rite(b64, prompt)
203
+ ritual_log(f"πŸœƒ Rite complete in {dt:.1f}s on GPU.")
204
  progress(1.0, desc="Done.")
205
 
206
+ meta = json.dumps(
207
+ {"rite": prompt, "gpu_seconds": round(dt, 1)}, indent=2
208
+ )
209
  return prophecy, meta, get_log()
210
  except requests.Timeout:
211
+ raise gr.Error("⏳ The oracle dawdled. Try a smaller image.")
212
  finally:
213
+ _busy.release()
 
214
 
215
 
216
  # ---------------------------------------------------------------- UI
 
220
  <span style="color:#39ff88;"> __ _-'-------------------------------.__</span>
221
  <span style="color:#39ff88;"> | | / P A D D L E O C R - V L \\</span>
222
  <span style="color:#39ff88;"> _- | | | C R Y P T D E C K |</span>
223
+ <span style="color:#39ff88;"> | | | | 1.6 GGUF β–Έ llama.cpp β–Έ ZERO-GPU RITES |</span>
224
  <span style="color:#39ff88;"> | | | |_________________________________________|</span>
225
  <span style="color:#39ff88;"> ____|____|-|_'---------____________________---.___________</span>
226
  <span style="color:#ff00cc;"> (o o o o o o o o o o o o o o o o o o o o)</span>
 
237
  footer {display: none !important;}
238
  #header {border: 1px solid #1e2a1e; border-radius: 10px;
239
  background: radial-gradient(ellipse at top, #0b1a10, #05060a); padding: 12px;}
 
240
  textarea, .output-html, .prose {font-family: 'Fira Mono', ui-monospace, monospace !important;}
241
  .rune-btn {font-family: monospace !important; text-transform: uppercase;
242
  letter-spacing: 2px; background: linear-gradient(90deg, #0f2b18, #14203a) !important;
 
247
  h1, h2, h3, .label, .info {font-family: ui-monospace, monospace !important;}
248
  """
249
 
250
+ with gr.Blocks(title="πŸ΄β€β˜ οΈ PaddleOCR-VL CryptDeck") as demo:
251
  gr.HTML(f'<div id="header">{HEADER}</div>')
252
 
253
  with gr.Row():
 
265
  )
266
  summon = gr.Button("πŸœ‚ SUMMON THE PROPHECY", elem_classes=["rune-btn"])
267
  gr.HTML(
268
+ '<div class="info" style="opacity:.75;">ZeroGPU rites: first cast hauls '
269
+ "~3 GB (engine + grimoire), then it's cached. Each rite boots the "
270
+ "oracle fresh onto the GPU (~10–20 s) before divining in seconds. "
271
  "One rite at a time, matey.</div>"
272
  )
273
  with gr.Column(scale=1):