File size: 2,929 Bytes
fd7251d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fa17859
fd7251d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
"""
Does this Space read a PDF the way the corpus was read?

This is the test the whole design rests on. Part A's index and Part B's F1 of 0.945 were measured
on `skeleton_masked` and `payload_window` strings produced by the EDA notebook's extractor. If
`corpus_text.py` produces even slightly different text, those numbers stop describing this app and
become decoration.

So it pulls real PDFs out of the generation repo, runs them through `corpus_text.py`, and compares
the result **character by character** against the published parquet. Not "close enough" — identical.

    python test_fidelity.py [n_files]
"""

import sys

import pandas as pd

import corpus_text

CORPUS = ("https://huggingface.co/datasets/Cyber-security-final-project/"
          "HARMLESS_Synthetic_Injected_PDFs_EDA/resolve/main/Datasets/"
          "synthetic_corpus_part2_clustered.parquet")
GENERATION_REPO = "Cyber-security-final-project/Generated_Injected_PDFs_HARMLESS"


def main(n=8):
    from huggingface_hub import hf_hub_download

    corpus = pd.read_parquet(CORPUS).set_index("file_id")

    # Injected and clean both, and not all from one family: masking and binary-stream handling
    # differ between them, and a test that only saw one would pass on a broken extractor.
    ids = list(corpus.index)
    sample = ids[::max(1, len(ids) // n)][:n]

    failures = 0
    for fid in sample:
        row = corpus.loc[fid]
        try:
            path = hf_hub_download(GENERATION_REPO, f"Output_PDFs/{fid}", repo_type="dataset")
        except Exception as e:
            print(f"  ? {fid}: could not fetch ({type(e).__name__})")
            continue

        skeleton, _, _ = corpus_text.build_skeleton(open(path, "rb").read())
        masked = corpus_text.mask_leaks(skeleton)
        window = corpus_text.payload_window(skeleton)

        checks = {"skeleton_masked": (masked, row["skeleton_masked"]),
                  "payload_window": (window, row["payload_window"])}

        # The single-window path must also be what the triage returns first for a marked file,
        # or the model reads a different string here than the corpus was scored on.
        top = corpus_text.candidate_windows(skeleton, cover_all=False)[0]

        bad = [k for k, (got, want) in checks.items() if got != want]
        if bad:
            failures += 1
            print(f"  x {fid}: differs in {', '.join(bad)}")
            for k in bad:
                got, want = checks[k]
                print(f"      {k}: got {len(got):,} chars, corpus has {len(want):,}")
        else:
            note = "head" if top["is_head"] else f"{len(top['families'])} family marker(s)"
            print(f"  . {fid}: identical  (triage top = {note})")

    print(f"\n{len(sample) - failures}/{len(sample)} identical to the published corpus")
    return 1 if failures else 0


if __name__ == "__main__":
    sys.exit(main(int(sys.argv[1]) if len(sys.argv) > 1 else 8))