Download nexora/data.py from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 3.55 kB
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/data.py
- Command line
-
hf download hf://devildasdf/NEXORA/nexora/data.py
-
curl -L -o data.py https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/data.py
3.55 kB
| """Auditable small-corpus pipeline. Quadratic near-dedup is intentionally lab-only.""" | |
| from collections import Counter | |
| from dataclasses import dataclass, asdict | |
| from hashlib import sha256 | |
| from pathlib import Path | |
| import json | |
| import re | |
| import unicodedata | |
| import numpy as np | |
| from .tokenizer import ByteTokenizer | |
| SECRET = re.compile(r"(?:hf_[A-Za-z0-9]{20,}|sk-[A-Za-z0-9_-]{20,}|-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----)") | |
| EMAIL = re.compile(r"\b[\w.+-]+@[\w.-]+\.[A-Za-z]{2,}\b") | |
| ALLOWED = {"MIT", "Apache-2.0", "BSD-3-Clause", "CC0-1.0", "owner-authored"} | |
| class Record: | |
| id: str | |
| text: str | |
| source: str | |
| license: str | |
| domain: str | |
| split: str = "train" | |
| def shingles(text, width=5): | |
| words = text.lower().split() | |
| return {tuple(words[i:i + width]) for i in range(max(1, len(words) - width + 1))} | |
| def similarity(a, b): | |
| return len(a & b) / max(1, len(a | b)) | |
| def prepare(records, out_dir, *, holdouts=(), near_threshold=0.9): | |
| if not 0 <= near_threshold <= 1: | |
| raise ValueError("Invalid near duplicate threshold") | |
| out = Path(out_dir) | |
| out.mkdir(parents=True, exist_ok=True) | |
| seen, signatures, accepted, reasons = set(), [], [], Counter() | |
| holdout_sigs = [shingles(unicodedata.normalize("NFC", s)) for s in holdouts] | |
| for raw in records: | |
| r = Record(**raw) if isinstance(raw, dict) else raw | |
| if r.license not in ALLOWED or not r.source or r.split not in {"train", "validation", "test"}: | |
| reasons["license_or_provenance_or_split"] += 1 | |
| continue | |
| text = unicodedata.normalize("NFC", r.text.replace("\r\n", "\n")) | |
| if SECRET.search(text): | |
| reasons["secret"] += 1 | |
| continue | |
| text = EMAIL.sub("[EMAIL]", text) | |
| if len(text.strip()) < 24 or "\x00" in text: | |
| reasons["quality"] += 1 | |
| continue | |
| digest = sha256(text.encode()).hexdigest() | |
| sig = shingles(text) | |
| if digest in seen: | |
| reasons["exact_duplicate"] += 1 | |
| continue | |
| if any(similarity(sig, s) >= near_threshold for s in holdout_sigs): | |
| reasons["contamination"] += 1 | |
| continue | |
| if any(similarity(sig, s) >= near_threshold for s in signatures): | |
| reasons["near_duplicate"] += 1 | |
| continue | |
| seen.add(digest) | |
| signatures.append(sig) | |
| accepted.append({**asdict(r), "text": text, "sha256": digest, | |
| "language_hint": "hi-or-mixed" if re.search("[\u0900-\u097f]", text) else "und", | |
| "quality_score": min(1.0, len(set(text)) / 40)}) | |
| tok, shards = ByteTokenizer(), {} | |
| for split in ("train", "validation", "test"): | |
| ids = [i for r in accepted if r["split"] == split for i in tok.encode(r["text"], special=True)] | |
| path = out / f"{split}.npy" | |
| np.save(path, np.asarray(ids, dtype=np.uint16), allow_pickle=False) | |
| shards[split] = {"file": path.name, "tokens": len(ids), "sha256": sha256(path.read_bytes()).hexdigest()} | |
| (out / "records.jsonl").write_text("".join(json.dumps(r, ensure_ascii=False) + "\n" for r in accepted), encoding="utf-8") | |
| manifest = {"version": 1, "tokenizer": "utf8-byte-v1", "accepted": len(accepted), | |
| "rejected": dict(reasons), "shards": shards, "limitations": ["PII regex is incomplete", "language hint is not language identification", "near dedup is quadratic"]} | |
| (out / "manifest.json").write_text(json.dumps(manifest, indent=2), encoding="utf-8") | |
| return manifest | |