Spaces:
Running on Zero
Running on Zero
File size: 16,126 Bytes
4a6ccb0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 | """
Stage 2 - asking MiMo.
The prompt, the prefill and the parser are **lifted verbatim from Part B**
(`Evaluation_of_OpenSource_Models_for_PDF_Injection_Recognition/Final_Project_Evaluation_V1.ipynb`,
cell 48). Same system message, same closed family list in the same order, same MiMo prefill, same
brace-counting parser, same greedy decoding, same 200-token cap. A reworded prompt or a widened
parser is a different experiment, and Part B's measured F1 of 0.945 would no longer describe it.
**One runtime: the GPU.** MiMo runs here as the original BF16 checkpoint quantised to 4-bit NF4 by
`bitsandbytes` and batched exactly as Part B batched it, which *is* Part B's configuration - nothing
about the arithmetic differs, so the published F1 of 0.945 describes this code.
An earlier version of this module also carried a `llama.cpp` path so the Space could fall back to
the CPU once a visitor's daily ZeroGPU quota ran out. It is gone, because it could never run here:
the prebuilt `llama-cpp-python` wheels are musl-linked while a Space is glibc, and compiling the
sdist exceeds the Space build timeout. Offering a runtime that only ever failed on being chosen was
worse than not offering it, so the choice - and the interface control that presented it - has been
removed rather than left as decoration.
"""
import json
import os
import re
import threading
FAMILIES = ["cross_site_scripting", "dde_template_injection", "javascript_injection",
"llm_prompt_injection", "object_action_injection", "polyglot_file",
"ransomware_simulation", "shellcode_embedded_exe", "ssrf",
"steganographic_payload", "uri_redirect_phishing", "xfa_acroform_injection"]
# ---------------------------------------------------------------------------------------------
# Verbatim from Part B, cell 48
# ---------------------------------------------------------------------------------------------
SYSTEM = (
"You are a PDF security analyst. You are given the raw extracted text of a PDF file - object "
"definitions, stream contents and metadata, exactly as they appear in the file. Some of these "
"files have had a malicious payload injected into them; most, but not all, have. Your job is to "
"say which, and to point at your evidence.\n\n"
"Answer with a single JSON object and nothing else:\n"
'{"injected": true or false, '
'"injection_type": one of ' + json.dumps(FAMILIES) + ' or "none", '
'"evidence": the exact substring from the input that convinced you, at most 200 characters, '
'or "" if none, '
'"reasoning": one short sentence}\n\n'
"If the file looks clean, answer injected=false and injection_type=\"none\". Do not guess a "
"family when you do not believe there is an injection."
)
MAX_CHARS = 3000 # payload_window is already capped at this; belt and braces
MAX_NEW = 200
BATCH = 8 # Part B's batch, quartered automatically on CUDA OOM
def build_messages(text: str):
"""The chat turns for one window."""
return [{"role": "system", "content": SYSTEM},
{"role": "user", "content": "PDF extract:\n\n```\n" + text[:MAX_CHARS] + "\n```"}]
# Text appended to the assistant turn, so the model resumes from it instead of starting free.
# MiMo is reasoning-trained, opens every answer with `<think>`, and at MAX_NEW = 200 the budget is
# gone before the block closes - not one of its 1,100 Part B answers contained a closing
# `</think>`, so it never reached the JSON. An empty, already-closed block says the deliberation is
# finished before it begins, and the opening brace puts it inside the answer.
PREFILL = '<think>\n\n</think>\n\n{"injected":'
def scan_objects(raw: str):
r"""
Every balanced {...} in the text, counting braces and skipping anything inside a string literal.
A regex cannot do this - matching balanced delimiters is outside what regular expressions can
express - and the naive `\{[^{}]*\}` this replaced was actively harmful: every injected file in
this corpus carries an EICAR-style marker containing a `}`, so the moment a model quoted its
evidence the match was truncated mid-string and the verdict was discarded. The bug fired
exactly when the model was RIGHT.
"""
objs, depth, start, in_str, esc = [], 0, None, False, False
for i, ch in enumerate(raw or ""):
if in_str:
if esc:
esc = False
elif ch == "\\":
esc = True
elif ch == '"':
in_str = False
continue
if ch == '"':
in_str = True
elif ch == "{":
if depth == 0:
start = i
depth += 1
elif ch == "}":
depth -= 1
if depth == 0 and start is not None:
objs.append(raw[start:i + 1])
start = None
depth = max(depth, 0)
return objs
def parse_response(raw: str) -> dict:
"""
Pull the verdict out of whatever the model said.
The LAST balanced object is taken, not the first: reasoning-trained models restate the schema
while thinking and emit the real answer at the end. If nothing parses as JSON the two fields
that matter are lifted out individually rather than thrown away. Only a response with no
recoverable verdict counts as parse_ok=False, and that reads as 'not injected': a detector that
cannot make itself understood has caught nothing.
Kept deliberately narrow, exactly as Part B scored it. A wider salvage would recover more
verdicts and would also mean the F1 quoted in the UI describes a parser that is not this one.
"""
for m in reversed(scan_objects(raw)):
try:
obj = json.loads(m)
except json.JSONDecodeError:
continue
if "injected" in obj:
inj = obj["injected"]
inj = inj if isinstance(inj, bool) else str(inj).strip().lower() in {"true", "yes", "1"}
fam = str(obj.get("injection_type", "none") or "none").strip().lower()
return {"parse_ok": True, "parsed_by": "balanced JSON",
"pred_injected": int(inj),
"pred_family": fam if fam in FAMILIES else "none",
"evidence": str(obj.get("evidence", ""))[:200],
"reasoning": str(obj.get("reasoning", ""))[:300]}
m = re.search(r'"injected"\s*:\s*(true|false)', raw or "", re.I)
if m:
f = re.search(r'"injection_type"\s*:\s*"([a-z_]+)"', raw, re.I)
fam = f.group(1).lower() if f else "none"
return {"parse_ok": True, "parsed_by": "field regex",
"pred_injected": int(m.group(1).lower() == "true"),
"pred_family": fam if fam in FAMILIES else "none",
"evidence": "", "reasoning": ""}
return {"parse_ok": False, "parsed_by": "unrecoverable", "pred_injected": 0,
"pred_family": "none", "evidence": "", "reasoning": ""}
# ---------------------------------------------------------------------------------------------
# The runtime
# ---------------------------------------------------------------------------------------------
BASE_REPO = "XiaomiMiMo/MiMo-7B-RL"
# HF sets this on a ZeroGPU Space. It is checked instead of `torch.cuda.is_available()` because on
# ZeroGPU there is no device at import time - one is attached only inside a @spaces.GPU call - so
# asking torch at startup would answer "no GPU" on the very hardware that has one.
ZERO_GPU = bool(os.environ.get("SPACES_ZERO_GPU"))
def _cuda_present() -> bool:
try:
import torch
return torch.cuda.is_available()
except Exception:
return False
GPU_PRESENT = ZERO_GPU or _cuda_present()
RUNTIME_NOTE = (
"This Space is running Part B's own configuration: the BF16 checkpoint of "
"XiaomiMiMo/MiMo-7B-RL quantised to 4-bit NF4 by bitsandbytes, greedy, 200 new tokens, "
"batched at 8. The prompt, prefill, decoding and parser are byte-identical to the run that "
"produced F1 0.945, so that figure describes this configuration - measured on the project's "
"own synthetic corpus, which is the thing it does not describe.")
# Measured in Part B on a T4.
SECONDS_PER_WINDOW = 4.2
_lock = threading.Lock() # one model, one scan at a time
_tokenizer = None
def load_tokenizer():
"""
MiMo's own tokenizer.
Part B let each model's own tokenizer apply the chat template - the wrapper of role tags around
the prompt - so doing the same is what keeps the rendered string identical to the one that was
measured.
"""
global _tokenizer
if _tokenizer is None:
from transformers import AutoTokenizer
_tokenizer = AutoTokenizer.from_pretrained(BASE_REPO, trust_remote_code=True)
return _tokenizer
def render_prompt(text: str):
"""
Apply MiMo's chat template to one window, then append the prefill.
If the tokenizer cannot be reached the ChatML fallback is used. MiMo is a ChatML model, so this
produces the same string in practice - but it is a reconstruction rather than the model's own
template, so the caller is told which route was taken instead of the difference being silent.
"""
messages = build_messages(text)
try:
rendered = load_tokenizer().apply_chat_template(
messages, tokenize=False, add_generation_prompt=True)
route = "tokenizer template"
except Exception:
rendered = "".join(f"<|im_start|>{m['role']}\n{m['content']}<|im_end|>\n"
for m in messages) + "<|im_start|>assistant\n"
route = "ChatML fallback"
return rendered + PREFILL, route
# ---------------------------------------------------------------------------------------------
# GPU path - Part B's configuration, unchanged
# ---------------------------------------------------------------------------------------------
_gpu_model = None
def _gpu_decorator(duration: int):
"""
`@spaces.GPU` on a ZeroGPU Space, and a no-op anywhere else.
ZeroGPU scans for a decorated function **at startup** and refuses to serve the Space if it
finds none, so the decoration has to happen at import time - which is why this is a decorator
factory applied below rather than a check made when a scan begins.
"""
try:
import spaces
return spaces.GPU(duration=duration)
except Exception:
return lambda fn: fn
def prefetch_weights(progress=None):
"""
Pull the checkpoint to local disk **before** any GPU is requested.
ZeroGPU bills wall-clock inside the decorated call and caps how long one may last, so a 15.7 GB
first-run download in there does not merely waste the grant - it guarantees the very first scan
exceeds the cap and fails. Downloading out here costs nothing but patience, and every later
call finds the files cached.
"""
from huggingface_hub import snapshot_download
if progress:
progress("fetching MiMo-7B weights (15.7 GB, first run only)")
snapshot_download(BASE_REPO, allow_patterns=["*.safetensors", "*.json", "*.txt", "*.model"])
def _load_gpu():
"""MiMo in 4-bit NF4 on the attached device. Cached; ~5 GB of weights."""
global _gpu_model
if _gpu_model is None:
import torch
from transformers import AutoModelForCausalLM, BitsAndBytesConfig
q = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4",
bnb_4bit_compute_dtype=torch.float16,
bnb_4bit_use_double_quant=True)
# device_map={"": 0} rather than "auto": everything on the one GPU or nothing. "auto" is
# free to spill layers onto the CPU, which is exactly the state bitsandbytes then refuses
# to work in, and the failure arrives as an offload complaint instead of "it does not fit".
_gpu_model = AutoModelForCausalLM.from_pretrained(
BASE_REPO, quantization_config=q, device_map={"": 0}, trust_remote_code=True)
_gpu_model.eval()
return _gpu_model
# Two separate limits squeeze this number, and the free one is much tighter than the hard one.
#
# The hard limit: the `spaces` client asks the scheduler for 1.5x whatever is set here, and ZeroGPU
# refuses anything over 300s - so above 200 is rejected outright, which is how the first deploy of
# this path died.
#
# The limit that actually bites: a free Hugging Face account gets roughly **five minutes of ZeroGPU
# per day**, and the scheduler reserves the full requested duration up front rather than what the
# run turns out to need. Asking for 180 (=270s reserved) therefore spends a whole day's quota on
# one batch. 110 covers the 4-bit load (~60s from cached weights) plus MAX_WINDOWS_GPU regions at
# ~4.2s each, reserves 165s, and leaves room for a second run in the same day.
GPU_DURATION = 110
MAX_WINDOWS_GPU = 8
@_gpu_decorator(GPU_DURATION)
def _judge_all_gpu(texts: list) -> list:
"""
Every window in one GPU call, batched the way Part B batched documents.
One call rather than one per window because ZeroGPU grants and reclaims the device around each
decorated call, and paying the model-load cost per window would dominate everything else.
"""
import torch
model = _load_gpu()
tok = load_tokenizer()
rendered = [render_prompt(t) for t in texts]
prompts = [p for p, _ in rendered]
route = rendered[0][1] if rendered else "tokenizer template"
tok.padding_side = "left" # decoder-only: right padding starts generation after it
if tok.pad_token is None:
tok.pad_token = tok.eos_token
out, i, batch = [], 0, BATCH
while i < len(prompts):
chunk = prompts[i:i + batch]
try:
enc = tok(chunk, return_tensors="pt", padding=True,
truncation=True, max_length=2048).to(model.device)
with torch.inference_mode():
gen = model.generate(**enc, max_new_tokens=MAX_NEW, do_sample=False,
pad_token_id=tok.pad_token_id)
decoded = tok.batch_decode(gen[:, enc["input_ids"].shape[1]:], skip_special_tokens=True)
except torch.cuda.OutOfMemoryError:
torch.cuda.empty_cache()
if batch == 1:
raise
batch = max(1, batch // 4) # Part B's policy: shrink rather than hard-code
continue
# generate() returns only the new tokens, so the prefill was given to the model but never
# generated. Put it back before parsing or the opening brace of the JSON is missing.
for text in decoded:
raw = PREFILL + text
out.append({**parse_response(raw), "raw": raw.strip()[:2000], "prompt_route": route})
i += len(chunk)
return out
# ---------------------------------------------------------------------------------------------
# The one entry point the app uses
# ---------------------------------------------------------------------------------------------
def judge_all(texts: list, progress=None) -> list:
"""
Windows in, one parsed verdict each, in order.
Everything goes in a single GPU call, so this cannot report progress part-way through the
generation itself - only around it. The lock is what stops two scans sharing one model.
"""
if not texts:
return []
if not GPU_PRESENT:
raise RuntimeError("this Space has no GPU; 4-bit loading through bitsandbytes needs CUDA")
with _lock:
prefetch_weights(progress) # outside the GPU grant, deliberately
if progress:
progress(f"MiMo reading {len(texts)} region(s)")
return _judge_all_gpu(texts)
|