Spaces:
Running on Zero
Running on Zero
| """ | |
| Stage 4a - the seventeen numbers that describe the shape of a document. | |
| These are the `document shape` block of the family-naming model's feature vector, and every one of | |
| them is **lifted verbatim from the corpus build** (`HARMLESS_Synthetic_Injected_PDFs_EDA/ | |
| Final_project_V7_EDA.ipynb`, cells 89-90, `extract_one`). That is the correctness argument for this | |
| file: the model was fitted on columns produced by that function, so a feature computed even | |
| slightly differently here is a different variable wearing the same name, and the model would score | |
| it confidently and wrongly. | |
| `corpus_text.py` already carries the text half of that same cell (the skeleton builder, the marker | |
| alternation). It is imported rather than copied, so the two halves cannot drift apart. | |
| Do not "improve" the arithmetic in this file. If a definition here looks odd - `n_obj` counting the | |
| string " obj" rather than parsing the xref table, `n_distinct_names` reading only the first 2 MB - | |
| that oddity is in the training data too, and reproducing it is the whole point. | |
| """ | |
| import re | |
| import numpy as np | |
| import corpus_text | |
| # EDA cell 89. `STREAM_RE`, `_printable_frac` and the skeleton builder live in `corpus_text`. | |
| NAME_RE = re.compile(rb"/[A-Za-z][A-Za-z0-9#]{1,30}") | |
| IMAGE_RE = re.compile(rb"/Subtype\s*/Image") | |
| BRACE_RE = re.compile(rb"<<|>>") | |
| # The order the model was fitted on. `family_model.py` slices by position, so this list is a | |
| # schema, not a convenience - it is asserted against the model's own spec file at import. | |
| NUMERIC = ["file_size", "n_obj", "n_stream", "n_objstm", "n_images", "n_distinct_names", | |
| "max_dict_depth", "printable_frac", "entropy_file", "entropy_largest_stream", | |
| "largest_stream_bytes", "compression_ratio", "binary_streams_dropped", | |
| "n_pages", "page_text_chars", "skeleton_chars", "was_truncated"] | |
| # Heavy-tailed byte counts got a log companion in training: trees do not need one, and the | |
| # logistic-regression baseline the model was compared against did. | |
| LOGGED = ["file_size", "largest_stream_bytes", "page_text_chars"] | |
| def entropy(chunk: bytes, sample: int = 1_000_000) -> float: | |
| """Shannon entropy in bits/byte. 8.0 = incompressible; high values mean packed or encrypted.""" | |
| if not chunk: | |
| return 0.0 | |
| counts = np.bincount(np.frombuffer(chunk[:sample], dtype=np.uint8), | |
| minlength=256).astype(np.float64) | |
| counts = counts[counts > 0] | |
| p = counts / counts.sum() | |
| return float(-(p * np.log2(p)).sum()) | |
| def max_dict_depth(data: bytes, limit: int = 2_000_000) -> int: | |
| """ | |
| Deepest nesting of PDF dictionaries (<< >>) - a cheap proxy for object complexity. | |
| Every '<<' is +1 and every '>>' is -1; the answer is the running maximum of the cumulative sum. | |
| """ | |
| steps = np.array([1 if m.group() == b"<<" else -1 | |
| for m in BRACE_RE.finditer(data[:limit])], dtype=np.int32) | |
| if steps.size == 0: | |
| return 0 | |
| return int(np.maximum.accumulate(np.cumsum(steps)).max()) | |
| def page_stats(path): | |
| """ | |
| Page count and visible text length, via PyMuPDF. | |
| These two describe the *carrier* document - the innocent PDF the payload was injected into - | |
| rather than the payload, which lives in the object structure and adds almost no visible text. | |
| They were controls in the EDA and they are ordinary features here. | |
| A malformed file is expected rather than exceptional in this corpus, so a parse failure returns | |
| zeros exactly as `extract_one` recorded them, instead of propagating. The training rows for | |
| unparseable files carried those same zeros. | |
| """ | |
| try: | |
| import fitz | |
| fitz.TOOLS.mupdf_display_errors(False) | |
| with fitz.open(path) as doc: | |
| return doc.page_count, sum(len(page.get_text()) for page in doc), True | |
| except Exception: | |
| return 0, 0, False | |
| def describe(path, data: bytes = None, skeleton: str = None, | |
| truncated: bool = None, dropped: int = None) -> dict: | |
| """ | |
| The seventeen numbers for one PDF, by the corpus's own definitions. | |
| The skeleton is passed in when the caller already built one - `app.py` does, on upload - because | |
| rendering a 780 KB PDF to text twice per scan is the single most expensive thing this module | |
| could do for no reason. | |
| """ | |
| if data is None: | |
| with open(path, "rb") as fh: | |
| data = fh.read() | |
| if skeleton is None: | |
| skeleton, truncated, dropped = corpus_text.build_skeleton(data) | |
| n = len(data) | |
| streams = [m.group(2) for m in corpus_text.STREAM_RE.finditer(data)] | |
| largest = max(streams, key=len) if streams else b"" | |
| n_pages, page_text_chars, parses_ok = page_stats(path) | |
| return { | |
| "file_size": n, | |
| "n_obj": data.count(b" obj"), | |
| "n_stream": len(streams), | |
| "n_objstm": data.count(b"/ObjStm"), | |
| "n_images": len(IMAGE_RE.findall(data)), | |
| "n_distinct_names": len(set(NAME_RE.findall(data[:2_000_000]))), | |
| "max_dict_depth": max_dict_depth(data), | |
| "printable_frac": corpus_text._printable_frac(data, sample=n), | |
| "entropy_file": entropy(data[:1_000_000]), | |
| "entropy_largest_stream": entropy(largest[:1_000_000]), | |
| "largest_stream_bytes": len(largest), | |
| "compression_ratio": len(skeleton) / n if n else np.nan, | |
| "binary_streams_dropped": int(dropped or 0), | |
| "n_pages": n_pages, | |
| "page_text_chars": page_text_chars, | |
| "skeleton_chars": len(skeleton), | |
| "was_truncated": bool(truncated), | |
| # Not a model feature. Carried so the interface can say the file did not parse, which is | |
| # worth knowing on its own and explains why the two page features are zero. | |
| "parses_ok": parses_ok, | |
| } | |
| def vector(stats: dict) -> np.ndarray: | |
| """The seventeen numbers plus their three log companions, in the fitted order. Shape (20,).""" | |
| base = np.array([float(stats[k]) for k in NUMERIC], dtype=np.float32) | |
| base = np.nan_to_num(base) | |
| logs = np.log1p(np.abs(np.array([float(stats[k]) for k in LOGGED], dtype=np.float32))) | |
| return np.concatenate([base, np.nan_to_num(logs)]) | |