"""Auditable small-corpus pipeline. Quadratic near-dedup is intentionally lab-only.""" from collections import Counter from dataclasses import dataclass, asdict from hashlib import sha256 from pathlib import Path import json import re import unicodedata import numpy as np from .tokenizer import ByteTokenizer SECRET = re.compile(r"(?:hf_[A-Za-z0-9]{20,}|sk-[A-Za-z0-9_-]{20,}|-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----)") EMAIL = re.compile(r"\b[\w.+-]+@[\w.-]+\.[A-Za-z]{2,}\b") ALLOWED = {"MIT", "Apache-2.0", "BSD-3-Clause", "CC0-1.0", "owner-authored"} @dataclass class Record: id: str text: str source: str license: str domain: str split: str = "train" def shingles(text, width=5): words = text.lower().split() return {tuple(words[i:i + width]) for i in range(max(1, len(words) - width + 1))} def similarity(a, b): return len(a & b) / max(1, len(a | b)) def prepare(records, out_dir, *, holdouts=(), near_threshold=0.9): if not 0 <= near_threshold <= 1: raise ValueError("Invalid near duplicate threshold") out = Path(out_dir) out.mkdir(parents=True, exist_ok=True) seen, signatures, accepted, reasons = set(), [], [], Counter() holdout_sigs = [shingles(unicodedata.normalize("NFC", s)) for s in holdouts] for raw in records: r = Record(**raw) if isinstance(raw, dict) else raw if r.license not in ALLOWED or not r.source or r.split not in {"train", "validation", "test"}: reasons["license_or_provenance_or_split"] += 1 continue text = unicodedata.normalize("NFC", r.text.replace("\r\n", "\n")) if SECRET.search(text): reasons["secret"] += 1 continue text = EMAIL.sub("[EMAIL]", text) if len(text.strip()) < 24 or "\x00" in text: reasons["quality"] += 1 continue digest = sha256(text.encode()).hexdigest() sig = shingles(text) if digest in seen: reasons["exact_duplicate"] += 1 continue if any(similarity(sig, s) >= near_threshold for s in holdout_sigs): reasons["contamination"] += 1 continue if any(similarity(sig, s) >= near_threshold for s in signatures): reasons["near_duplicate"] += 1 continue seen.add(digest) signatures.append(sig) accepted.append({**asdict(r), "text": text, "sha256": digest, "language_hint": "hi-or-mixed" if re.search("[\u0900-\u097f]", text) else "und", "quality_score": min(1.0, len(set(text)) / 40)}) tok, shards = ByteTokenizer(), {} for split in ("train", "validation", "test"): ids = [i for r in accepted if r["split"] == split for i in tok.encode(r["text"], special=True)] path = out / f"{split}.npy" np.save(path, np.asarray(ids, dtype=np.uint16), allow_pickle=False) shards[split] = {"file": path.name, "tokens": len(ids), "sha256": sha256(path.read_bytes()).hexdigest()} (out / "records.jsonl").write_text("".join(json.dumps(r, ensure_ascii=False) + "\n" for r in accepted), encoding="utf-8") manifest = {"version": 1, "tokenizer": "utf8-byte-v1", "accepted": len(accepted), "rejected": dict(reasons), "shards": shards, "limitations": ["PII regex is incomplete", "language hint is not language identification", "near dedup is quadratic"]} (out / "manifest.json").write_text(json.dumps(manifest, indent=2), encoding="utf-8") return manifest