File size: 6,147 Bytes
4a6ccb0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
"""
Stage 4a - the seventeen numbers that describe the shape of a document.

These are the `document shape` block of the family-naming model's feature vector, and every one of
them is **lifted verbatim from the corpus build** (`HARMLESS_Synthetic_Injected_PDFs_EDA/
Final_project_V7_EDA.ipynb`, cells 89-90, `extract_one`). That is the correctness argument for this
file: the model was fitted on columns produced by that function, so a feature computed even
slightly differently here is a different variable wearing the same name, and the model would score
it confidently and wrongly.

`corpus_text.py` already carries the text half of that same cell (the skeleton builder, the marker
alternation). It is imported rather than copied, so the two halves cannot drift apart.

Do not "improve" the arithmetic in this file. If a definition here looks odd - `n_obj` counting the
string " obj" rather than parsing the xref table, `n_distinct_names` reading only the first 2 MB -
that oddity is in the training data too, and reproducing it is the whole point.
"""

import re

import numpy as np

import corpus_text

# EDA cell 89. `STREAM_RE`, `_printable_frac` and the skeleton builder live in `corpus_text`.
NAME_RE = re.compile(rb"/[A-Za-z][A-Za-z0-9#]{1,30}")
IMAGE_RE = re.compile(rb"/Subtype\s*/Image")
BRACE_RE = re.compile(rb"<<|>>")

# The order the model was fitted on. `family_model.py` slices by position, so this list is a
# schema, not a convenience - it is asserted against the model's own spec file at import.
NUMERIC = ["file_size", "n_obj", "n_stream", "n_objstm", "n_images", "n_distinct_names",
           "max_dict_depth", "printable_frac", "entropy_file", "entropy_largest_stream",
           "largest_stream_bytes", "compression_ratio", "binary_streams_dropped",
           "n_pages", "page_text_chars", "skeleton_chars", "was_truncated"]

# Heavy-tailed byte counts got a log companion in training: trees do not need one, and the
# logistic-regression baseline the model was compared against did.
LOGGED = ["file_size", "largest_stream_bytes", "page_text_chars"]


def entropy(chunk: bytes, sample: int = 1_000_000) -> float:
    """Shannon entropy in bits/byte. 8.0 = incompressible; high values mean packed or encrypted."""
    if not chunk:
        return 0.0
    counts = np.bincount(np.frombuffer(chunk[:sample], dtype=np.uint8),
                         minlength=256).astype(np.float64)
    counts = counts[counts > 0]
    p = counts / counts.sum()
    return float(-(p * np.log2(p)).sum())


def max_dict_depth(data: bytes, limit: int = 2_000_000) -> int:
    """
    Deepest nesting of PDF dictionaries (<< >>) - a cheap proxy for object complexity.

    Every '<<' is +1 and every '>>' is -1; the answer is the running maximum of the cumulative sum.
    """
    steps = np.array([1 if m.group() == b"<<" else -1
                      for m in BRACE_RE.finditer(data[:limit])], dtype=np.int32)
    if steps.size == 0:
        return 0
    return int(np.maximum.accumulate(np.cumsum(steps)).max())


def page_stats(path):
    """
    Page count and visible text length, via PyMuPDF.

    These two describe the *carrier* document - the innocent PDF the payload was injected into -
    rather than the payload, which lives in the object structure and adds almost no visible text.
    They were controls in the EDA and they are ordinary features here.

    A malformed file is expected rather than exceptional in this corpus, so a parse failure returns
    zeros exactly as `extract_one` recorded them, instead of propagating. The training rows for
    unparseable files carried those same zeros.
    """
    try:
        import fitz
        fitz.TOOLS.mupdf_display_errors(False)
        with fitz.open(path) as doc:
            return doc.page_count, sum(len(page.get_text()) for page in doc), True
    except Exception:
        return 0, 0, False


def describe(path, data: bytes = None, skeleton: str = None,
             truncated: bool = None, dropped: int = None) -> dict:
    """
    The seventeen numbers for one PDF, by the corpus's own definitions.

    The skeleton is passed in when the caller already built one - `app.py` does, on upload - because
    rendering a 780 KB PDF to text twice per scan is the single most expensive thing this module
    could do for no reason.
    """
    if data is None:
        with open(path, "rb") as fh:
            data = fh.read()
    if skeleton is None:
        skeleton, truncated, dropped = corpus_text.build_skeleton(data)

    n = len(data)
    streams = [m.group(2) for m in corpus_text.STREAM_RE.finditer(data)]
    largest = max(streams, key=len) if streams else b""
    n_pages, page_text_chars, parses_ok = page_stats(path)

    return {
        "file_size": n,
        "n_obj": data.count(b" obj"),
        "n_stream": len(streams),
        "n_objstm": data.count(b"/ObjStm"),
        "n_images": len(IMAGE_RE.findall(data)),
        "n_distinct_names": len(set(NAME_RE.findall(data[:2_000_000]))),
        "max_dict_depth": max_dict_depth(data),
        "printable_frac": corpus_text._printable_frac(data, sample=n),
        "entropy_file": entropy(data[:1_000_000]),
        "entropy_largest_stream": entropy(largest[:1_000_000]),
        "largest_stream_bytes": len(largest),
        "compression_ratio": len(skeleton) / n if n else np.nan,
        "binary_streams_dropped": int(dropped or 0),
        "n_pages": n_pages,
        "page_text_chars": page_text_chars,
        "skeleton_chars": len(skeleton),
        "was_truncated": bool(truncated),
        # Not a model feature. Carried so the interface can say the file did not parse, which is
        # worth knowing on its own and explains why the two page features are zero.
        "parses_ok": parses_ok,
    }


def vector(stats: dict) -> np.ndarray:
    """The seventeen numbers plus their three log companions, in the fitted order. Shape (20,)."""
    base = np.array([float(stats[k]) for k in NUMERIC], dtype=np.float32)
    base = np.nan_to_num(base)
    logs = np.log1p(np.abs(np.array([float(stats[k]) for k in LOGGED], dtype=np.float32)))
    return np.concatenate([base, np.nan_to_num(logs)])