File size: 3,546 Bytes
12496fc | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 | """Auditable small-corpus pipeline. Quadratic near-dedup is intentionally lab-only."""
from collections import Counter
from dataclasses import dataclass, asdict
from hashlib import sha256
from pathlib import Path
import json
import re
import unicodedata
import numpy as np
from .tokenizer import ByteTokenizer
SECRET = re.compile(r"(?:hf_[A-Za-z0-9]{20,}|sk-[A-Za-z0-9_-]{20,}|-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----)")
EMAIL = re.compile(r"\b[\w.+-]+@[\w.-]+\.[A-Za-z]{2,}\b")
ALLOWED = {"MIT", "Apache-2.0", "BSD-3-Clause", "CC0-1.0", "owner-authored"}
@dataclass
class Record:
id: str
text: str
source: str
license: str
domain: str
split: str = "train"
def shingles(text, width=5):
words = text.lower().split()
return {tuple(words[i:i + width]) for i in range(max(1, len(words) - width + 1))}
def similarity(a, b):
return len(a & b) / max(1, len(a | b))
def prepare(records, out_dir, *, holdouts=(), near_threshold=0.9):
if not 0 <= near_threshold <= 1:
raise ValueError("Invalid near duplicate threshold")
out = Path(out_dir)
out.mkdir(parents=True, exist_ok=True)
seen, signatures, accepted, reasons = set(), [], [], Counter()
holdout_sigs = [shingles(unicodedata.normalize("NFC", s)) for s in holdouts]
for raw in records:
r = Record(**raw) if isinstance(raw, dict) else raw
if r.license not in ALLOWED or not r.source or r.split not in {"train", "validation", "test"}:
reasons["license_or_provenance_or_split"] += 1
continue
text = unicodedata.normalize("NFC", r.text.replace("\r\n", "\n"))
if SECRET.search(text):
reasons["secret"] += 1
continue
text = EMAIL.sub("[EMAIL]", text)
if len(text.strip()) < 24 or "\x00" in text:
reasons["quality"] += 1
continue
digest = sha256(text.encode()).hexdigest()
sig = shingles(text)
if digest in seen:
reasons["exact_duplicate"] += 1
continue
if any(similarity(sig, s) >= near_threshold for s in holdout_sigs):
reasons["contamination"] += 1
continue
if any(similarity(sig, s) >= near_threshold for s in signatures):
reasons["near_duplicate"] += 1
continue
seen.add(digest)
signatures.append(sig)
accepted.append({**asdict(r), "text": text, "sha256": digest,
"language_hint": "hi-or-mixed" if re.search("[\u0900-\u097f]", text) else "und",
"quality_score": min(1.0, len(set(text)) / 40)})
tok, shards = ByteTokenizer(), {}
for split in ("train", "validation", "test"):
ids = [i for r in accepted if r["split"] == split for i in tok.encode(r["text"], special=True)]
path = out / f"{split}.npy"
np.save(path, np.asarray(ids, dtype=np.uint16), allow_pickle=False)
shards[split] = {"file": path.name, "tokens": len(ids), "sha256": sha256(path.read_bytes()).hexdigest()}
(out / "records.jsonl").write_text("".join(json.dumps(r, ensure_ascii=False) + "\n" for r in accepted), encoding="utf-8")
manifest = {"version": 1, "tokenizer": "utf8-byte-v1", "accepted": len(accepted),
"rejected": dict(reasons), "shards": shards, "limitations": ["PII regex is incomplete", "language hint is not language identification", "near dedup is quadratic"]}
(out / "manifest.json").write_text(json.dumps(manifest, indent=2), encoding="utf-8")
return manifest
|