Spaces:
Running on Zero
Running on Zero
File size: 13,273 Bytes
4a6ccb0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 | """
Stage 1 - turning an uploaded PDF into the exact kind of string the corpus was built from.
Everything in the first block is **lifted verbatim from the EDA notebook that built the corpus**
(`HARMLESS_Synthetic_Injected_PDFs_EDA/Final_project_V7_EDA.ipynb`, cells 88-90 and 107). That is
not tidiness, it is the correctness argument for this whole Space: Part A's embedding index and
Part B's F1 of 0.945 were both measured on `payload_window` strings produced by exactly this code.
Extract the text even slightly differently and those published numbers stop describing this app.
Do not "improve" anything above the `Triage` heading.
"""
import re
import zlib
import numpy as np
# ---------------------------------------------------------------------------------------------
# Verbatim from the corpus build - EDA cell 88
# ---------------------------------------------------------------------------------------------
STREAM_RE = re.compile(rb"(stream\r?\n)(.*?)(endstream)", re.S)
INJECTION_MARKERS = {
"javascript_injection": [rb"/S\s*/JavaScript", rb"/JS\s*\("],
"cross_site_scripting": [rb"<script>", rb"fetch\("],
"ssrf": [rb"169\.254\.169\.254", rb"/latest/meta-data"],
"object_action_injection": [rb"/S\s*/Launch", rb"cmd\.exe"],
"llm_prompt_injection": [rb"IGNORE PREVIOUS INSTRUCTIONS", rb"LLM-INJECT"],
"shellcode_embedded_exe": [rb"application#2Fx-msdownload", rb"MZ.{0,20}\\x90\\x90"],
"polyglot_file": [rb"PK\\x03\\x04", rb"POLYGLOT ZIP\+PDF"],
"dde_template_injection": [rb"remote\.dotm\?dde=", rb"=cmd\|"],
"steganographic_payload": [rb"LSB-STEGO"],
"ransomware_simulation": [rb"RANSIM TEST"],
"xfa_acroform_injection": [rb"/Type\s*/XFA", rb"xfa:data"],
"uri_redirect_phishing": [rb"phishing\.\w+\.test"],
}
FRAMEWORK_MARKERS = ["AMTSO", "WICAR", "OWASP", "AtomicRedTeam", "Metasploit",
"Glasswall", "mindcrypt", "RanSim", "RANSIM", "custom"]
BASE_PAYLOAD_MARKERS = ["EICAR-STANDARD-ANTIVIRUS-TEST-FILE", "AMTSO-POTENTIALLY-UNWANTED-TEST-FILE",
"WICAR-BROWSER-TEST-PAYLOAD", "RANSIM-KNOWBE4-ENCRYPTION-SIMULATION"]
# Brand / test-file identifiers a model could string-match on without reasoning about PDF
# structure. The structural markers above are deliberately NOT masked: those ARE the attack, not a
# giveaway label. Masking them would delete the thing the model is supposed to find.
LEAK_STRINGS = sorted(
set(FRAMEWORK_MARKERS) | set(BASE_PAYLOAD_MARKERS)
| {"EICAR", "RANSIM", "eicar-standard-antivirus-test-file",
r"X5O!P%@AP[4\PZX54(P^)7CC)7}$"},
key=len, reverse=True) # longest first: "EICAR-STANDARD-..." masks before "EICAR"
LEAK_RE = re.compile("|".join(re.escape(s) for s in LEAK_STRINGS), re.IGNORECASE)
# ---------------------------------------------------------------------------------------------
# Verbatim from the corpus build - EDA cell 89
# ---------------------------------------------------------------------------------------------
STREAM_BODY_CAP = 4_096 # chars kept from any single stream body
SKELETON_CHAR_BUDGET = 120_000 # ~30k tokens
HEAD_SHARE = 0.45 # of the budget; the rest is the tail window
SCAN_WINDOW = 8_000_000 # bytes scanned from each end for markers
CLEAN_RE = re.compile(r"[^\x20-\x7e\n]")
SPACES_RE = re.compile(r"[ ]{4,}")
def _printable_frac(chunk: bytes, sample: int = 200_000) -> float:
"""Fraction of bytes that are ordinary printable ASCII."""
if not chunk:
return 1.0
arr = np.frombuffer(chunk[:sample], dtype=np.uint8)
ok = ((arr >= 32) & (arr < 127)) | (arr == 9) | (arr == 10) | (arr == 13)
return float(ok.mean())
def build_skeleton(data: bytes):
"""Render a PDF as payload-preserving text. Returns (skeleton, was_truncated, n_binary_dropped)."""
dropped = 0
def replace(match):
nonlocal dropped
opener, body, closer = match.group(1), match.group(2), match.group(3)
try: # most streams are FlateDecode
inflated = zlib.decompress(body)
if _printable_frac(inflated) > 0.6:
return opener + inflated[:STREAM_BODY_CAP] + b"\n" + closer
except zlib.error:
pass
if _printable_frac(body) > 0.6: # already plain text
return opener + body[:STREAM_BODY_CAP] + closer
dropped += 1 # genuinely binary (an image)
return opener + b"<<BINARY %d bytes>>" % len(body) + closer
text = STREAM_RE.sub(replace, data).decode("latin-1")
# Truncate FIRST, then clean: no point running two character-class substitutions over 76
# million characters only to discard 99.8% of the result.
truncated = len(text) > SKELETON_CHAR_BUDGET
if truncated:
# Head AND tail. The two insertion strategies put payloads at opposite ends of the file,
# so a plain head truncation would lose most of them.
head_n = int(SKELETON_CHAR_BUDGET * HEAD_SHARE)
tail_n = SKELETON_CHAR_BUDGET - head_n
elided = len(text) - SKELETON_CHAR_BUDGET
text = (text[:head_n * 2]
+ f"\n<<... {elided} characters elided ...>>\n"
+ text[-tail_n * 2:])
text = CLEAN_RE.sub(" ", text) # drop control/binary residue
text = SPACES_RE.sub(" ", text)
if len(text) > SKELETON_CHAR_BUDGET: # enforce the budget after cleanup
head_n = int(SKELETON_CHAR_BUDGET * HEAD_SHARE)
text = text[:head_n] + text[-(SKELETON_CHAR_BUDGET - head_n):]
truncated = True
return text, truncated, dropped
def mask_leaks(skeleton: str) -> str:
"""Blank the brand identifiers, keep the structural shape of the injection intact."""
return LEAK_RE.sub(lambda m: "X" * len(m.group(0)), skeleton)
def scan_window(data: bytes, window: int = SCAN_WINDOW) -> bytes:
"""Head and tail of a file, for the regex marker scans."""
if len(data) <= 2 * window:
return data
return data[:window] + data[-window:]
def detect_markers(blob) -> list:
"""Which injection types are structurally present. Works on raw bytes or on a skeleton."""
data = blob.encode("latin-1", errors="replace") if isinstance(blob, str) else blob
data = scan_window(data)
return [name for name, patterns in INJECTION_MARKERS.items()
if any(re.search(p, data, re.IGNORECASE) for p in patterns)]
# One alternation instead of 26 separate scans - EDA cell 107. The earliest match of the union is
# by definition the earliest match of any individual pattern.
ANY_MARKER_RE = re.compile(
b"|".join([p for ps in INJECTION_MARKERS.values() for p in ps]
+ [re.escape(s).encode() for s in FRAMEWORK_MARKERS + BASE_PAYLOAD_MARKERS]),
re.IGNORECASE)
WINDOW = 1_500 # characters kept either side of the payload - EDA cell 107
def payload_window(text: str, half: int = WINDOW) -> str:
"""
The neighbourhood of the injection, exactly as the corpus column of this name was built.
Clean files have no marker, so they fall back to the head of the document - which keeps them
comparable in length rather than empty. Kept here unchanged because it is the single-window
case, and because `candidate_windows` below must agree with it character for character.
"""
m = ANY_MARKER_RE.search(text.encode("latin-1", errors="replace"))
if not m:
return text[:2 * half]
return text[max(0, m.start() - half): m.start() + half]
# ---------------------------------------------------------------------------------------------
# Triage - new here, and the one place this Space departs from the notebooks
# ---------------------------------------------------------------------------------------------
#
# The notebooks scored one window per document, because they already knew where the payload was.
# An uploaded file offers no such promise: a payload can sit anywhere, and on the free CPU tier
# MiMo reads roughly one window every couple of minutes, so "score every window" is not on offer.
#
# So the same marker alternation that located the corpus payload is run over the *whole* skeleton
# instead of stopping at the first hit. Every match becomes a candidate window with the identical
# +/-1,500-character shape, they are merged where they overlap, and the most marker-dense ones go
# to the model first. A file with no marker anywhere yields exactly one candidate - the head of
# the document - which is byte-identical to what `payload_window` returns for a clean corpus file.
#
# What this is NOT: a detector. The ranking decides reading order, never the verdict. It also only
# knows the twelve families' signatures, so a payload shaped like none of them is triaged as if it
# were clean and the model sees the head of the file. That limit is stated in the UI, not buried.
MAX_MATCH_SCAN = 4_000 # matches considered; a pathological file will not run forever
def marker_windows(skeleton: str, half: int = WINDOW) -> list:
"""
Every +/-`half` neighbourhood around a marker in `skeleton`, merged and ranked.
Returns dicts with `start`, `end`, `text`, `n_markers`, `families` and `is_head`, most
marker-dense first. Never empty: with no markers at all it returns the head window.
"""
blob = skeleton.encode("latin-1", errors="replace")
spans = []
for i, m in enumerate(ANY_MARKER_RE.finditer(blob)):
if i >= MAX_MATCH_SCAN:
break
spans.append((max(0, m.start() - half), m.start() + half, m.start()))
if not spans:
head = skeleton[:2 * half]
return [{"start": 0, "end": len(head), "text": head, "n_markers": 0,
"families": [], "is_head": True, "source": "head"}]
# Merge overlaps so two markers 200 characters apart are read once, not twice.
merged = []
for start, end, hit in spans:
if merged and start <= merged[-1]["end"]:
merged[-1]["end"] = max(merged[-1]["end"], end)
merged[-1]["hits"].append(hit)
else:
merged.append({"start": start, "end": end, "hits": [hit]})
out = []
for span in merged:
# A merged span can grow past one window. The model reads at most 2*half characters, so
# centre the slice on the first marker in the span rather than sending a longer string
# than any corpus row ever carried.
first = span["hits"][0]
start = max(0, first - half)
text = skeleton[start:start + 2 * half]
out.append({"start": start, "end": start + len(text), "text": text,
"n_markers": len(span["hits"]),
"families": detect_markers(text), "is_head": False, "source": "marker"})
# Density first, then position: an early hit is where the head-insertion strategy puts things.
out.sort(key=lambda w: (-len(w["families"]), -w["n_markers"], w["start"]))
return out
def candidate_windows(skeleton: str, half: int = WINDOW, cover_all: bool = True) -> list:
"""
Every region of the document worth sending to the model, in reading order.
Marker neighbourhoods come first, ranked by density - they are the likeliest place to find
something and the first batch should be the one worth spending. **Then the rest of the
document follows**, tiled into windows of the same size, in document order.
That tail is not padding. The marker alternation only knows the twelve families this project
generated, so a payload shaped like none of them produces no marker at all and, without the
sweep, would sit in a part of the file the model never saw while the report said "clean". With
it, the batches cover the whole skeleton and "unread" means genuinely unread rather than
unreachable.
`cover_all=False` gives the marker regions alone, which is what the corpus itself was built
from and what `test_fidelity.py` checks against.
"""
windows = marker_windows(skeleton, half)
if not cover_all:
return windows
size = 2 * half
covered = [(w["start"], w["end"]) for w in windows]
sweep = []
for start in range(0, max(len(skeleton), 1), size):
text = skeleton[start:start + size]
if not text.strip():
continue
# Skip a tile that a marker window already mostly covers, so the same text is not paid for
# twice. Half the tile is the threshold: less than that and there is unseen text in it.
overlap = sum(max(0, min(start + size, e) - max(start, s)) for s, e in covered)
if overlap >= len(text) / 2:
continue
sweep.append({"start": start, "end": start + len(text), "text": text, "n_markers": 0,
"families": detect_markers(text), "is_head": start == 0,
"source": "sweep"})
return windows + sweep
|