Spaces:
Running on Zero
Running on Zero
| """ | |
| Does this Space read a PDF the way the corpus was read? | |
| This is the test the whole design rests on. Part A's index and Part B's F1 of 0.945 were measured | |
| on `skeleton_masked` and `payload_window` strings produced by the EDA notebook's extractor. If | |
| `corpus_text.py` produces even slightly different text, those numbers stop describing this app and | |
| become decoration. | |
| So it pulls real PDFs out of the generation repo, runs them through `corpus_text.py`, and compares | |
| the result **character by character** against the published parquet. Not "close enough" — identical. | |
| python test_fidelity.py [n_files] | |
| """ | |
| import sys | |
| import pandas as pd | |
| import corpus_text | |
| CORPUS = ("https://huggingface.co/datasets/Cyber-security-final-project/" | |
| "HARMLESS_Synthetic_Injected_PDFs_EDA/resolve/main/Datasets/" | |
| "synthetic_corpus_part2_clustered.parquet") | |
| GENERATION_REPO = "Cyber-security-final-project/Generated_Injected_PDFs_HARMLESS" | |
| def main(n=8): | |
| from huggingface_hub import hf_hub_download | |
| corpus = pd.read_parquet(CORPUS).set_index("file_id") | |
| # Injected and clean both, and not all from one family: masking and binary-stream handling | |
| # differ between them, and a test that only saw one would pass on a broken extractor. | |
| ids = list(corpus.index) | |
| sample = ids[::max(1, len(ids) // n)][:n] | |
| failures = 0 | |
| for fid in sample: | |
| row = corpus.loc[fid] | |
| try: | |
| path = hf_hub_download(GENERATION_REPO, f"Output_PDFs/{fid}", repo_type="dataset") | |
| except Exception as e: | |
| print(f" ? {fid}: could not fetch ({type(e).__name__})") | |
| continue | |
| skeleton, _, _ = corpus_text.build_skeleton(open(path, "rb").read()) | |
| masked = corpus_text.mask_leaks(skeleton) | |
| window = corpus_text.payload_window(skeleton) | |
| checks = {"skeleton_masked": (masked, row["skeleton_masked"]), | |
| "payload_window": (window, row["payload_window"])} | |
| # The single-window path must also be what the triage returns first for a marked file, | |
| # or the model reads a different string here than the corpus was scored on. | |
| top = corpus_text.candidate_windows(skeleton, cover_all=False)[0] | |
| bad = [k for k, (got, want) in checks.items() if got != want] | |
| if bad: | |
| failures += 1 | |
| print(f" x {fid}: differs in {', '.join(bad)}") | |
| for k in bad: | |
| got, want = checks[k] | |
| print(f" {k}: got {len(got):,} chars, corpus has {len(want):,}") | |
| else: | |
| note = "head" if top["is_head"] else f"{len(top['families'])} family marker(s)" | |
| print(f" . {fid}: identical (triage top = {note})") | |
| print(f"\n{len(sample) - failures}/{len(sample)} identical to the published corpus") | |
| return 1 if failures else 0 | |
| if __name__ == "__main__": | |
| sys.exit(main(int(sys.argv[1]) if len(sys.argv) > 1 else 8)) | |