File size: 7,843 Bytes
3a559ac | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 | #!/usr/bin/env python3
"""Build the reproducible, source-level OOD set from a Tatoeba export.
The export is intentionally not committed. The resulting JSONL contains the
selected text plus source IDs and hashes so the published set can be audited.
"""
import argparse
import hashlib
import json
import tarfile
import urllib.request
from collections import Counter, defaultdict
from datetime import date
from pathlib import Path
SOURCE_URL = "https://downloads.tatoeba.org/exports/sentences.tar.bz2"
SOURCE_SHA256 = "8117f33886e94c3d371f1f13d0efa120a8172d67eb7006b2656c444c012645d0"
SOURCE_RELEASE = "2026-07-18 export (archive last-modified date)"
TOTAL_PER_CLASS = 8772
LANGUAGES = {
"afr": "Afrikaans", "ara": "Arabic", "asm": "Assamese", "ben": "Bengali",
"bul": "Bulgarian", "cat": "Catalan", "ces": "Czech", "cmn": "Mandarin Chinese",
"dan": "Danish", "deu": "German", "ell": "Greek", "epo": "Esperanto",
"eus": "Basque", "fin": "Finnish", "fra": "French", "gle": "Irish",
"glg": "Galician", "hau": "Hausa", "heb": "Hebrew", "hin": "Hindi",
"hrv": "Croatian", "hun": "Hungarian", "hye": "Armenian", "ind": "Indonesian",
"isl": "Icelandic", "ita": "Italian", "jpn": "Japanese", "kor": "Korean",
"lit": "Lithuanian", "lvs": "Latvian", "mar": "Marathi", "mkd": "Macedonian",
"mon": "Mongolian", "nld": "Dutch", "nno": "Norwegian Nynorsk",
"nob": "Norwegian Bokmal", "pol": "Polish", "por": "Portuguese",
"ron": "Romanian", "rus": "Russian", "slk": "Slovak", "slv": "Slovenian",
"spa": "Spanish", "srp": "Serbian", "swe": "Swedish", "swh": "Swahili",
"tat": "Tatar", "tha": "Thai", "tur": "Turkish", "ukr": "Ukrainian",
"vie": "Vietnamese", "yid": "Yiddish",
}
def normalize(text):
return " ".join(text.split()).strip()
def digest(text):
return hashlib.sha256(text.encode("utf-8")).hexdigest()
def bucket(text):
size = len(text.encode("utf-8"))
if size <= 32:
return "short"
if size <= 96:
return "medium"
return "long"
def add_candidate(heaps, lang, capacity, source_id, text):
"""Keep the lexicographically smallest stable hashes per language."""
import heapq
text_hash = digest(text)
key = (text_hash, str(source_id))
heap = heaps[lang]
item = (key[0], key[1], text)
if len(heap) < capacity:
heapq.heappush(heap, (-int(key[0], 16), key[1], text))
elif int(key[0], 16) < -heap[0][0]:
heapq.heapreplace(heap, (-int(key[0], 16), key[1], text))
def download(path):
if path.exists():
return
path.parent.mkdir(parents=True, exist_ok=True)
urllib.request.urlretrieve(SOURCE_URL, path)
def sha256_file(path):
h = hashlib.sha256()
with path.open("rb") as f:
for block in iter(lambda: f.read(1024 * 1024), b""):
h.update(block)
return h.hexdigest()
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--source", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--manifest", type=Path, required=True)
args = parser.parse_args()
actual_sha256 = sha256_file(args.source)
if actual_sha256 != SOURCE_SHA256:
raise SystemExit(f"source checksum mismatch: {actual_sha256} != {SOURCE_SHA256}")
targets = {"eng": TOTAL_PER_CLASS}
non_english = sorted(LANGUAGES)
base, remainder = divmod(TOTAL_PER_CLASS, len(non_english))
targets.update({lang: base + (i < remainder) for i, lang in enumerate(non_english)})
heaps = defaultdict(list)
seen_rows = 0
with tarfile.open(args.source, "r:bz2") as archive:
member = archive.extractfile("sentences.csv")
if member is None:
raise SystemExit("sentences.csv missing from source archive")
for raw in member:
seen_rows += 1
if seen_rows == 1:
continue
parts = raw.decode("utf-8").rstrip("\n").split("\t", 2)
if len(parts) != 3:
continue
source_id, lang, raw_text = parts
if lang not in targets:
continue
text = normalize(raw_text)
encoded = text.encode("utf-8")
if not 3 <= len(text) <= 276 or len(encoded) > 256:
continue
if not any(ch.isalpha() for ch in text):
continue
add_candidate(heaps, lang, targets[lang] * 4, source_id, text)
rows = []
used_text_hashes = set()
for lang in ["eng", *non_english]:
candidates = sorted(heaps[lang], key=lambda item: (item[0] * -1, item[1]))
label = "EN" if lang == "eng" else "NOT-EN"
selected = 0
for _, source_id, text in candidates:
text_hash = digest(text)
if text_hash in used_text_hashes:
continue
used_text_hashes.add(text_hash)
rows.append({
"id": f"tatoeba-{source_id}-{lang}",
"text": text,
"label": label,
"language": "en" if lang == "eng" else lang,
"language_name": "English" if lang == "eng" else LANGUAGES[lang],
"category": bucket(text),
"source": "Tatoeba",
"source_release": SOURCE_RELEASE,
"source_id": source_id,
"source_url": f"https://tatoeba.org/en/sentences/show/{source_id}",
"text_sha256": digest(text),
"utf8_bytes": len(text.encode("utf-8")),
})
selected += 1
if selected == targets[lang]:
break
if selected != targets[lang]:
raise SystemExit(f"not enough unique rows for {lang}: {selected}/{targets[lang]}")
if len(rows) != TOTAL_PER_CLASS * 2:
raise SystemExit(f"wrong row count: {len(rows)}")
if len({row["text_sha256"] for row in rows}) != len(rows):
raise SystemExit("duplicate normalized text hashes found")
counts = Counter(row["language"] for row in rows)
if counts["en"] != TOTAL_PER_CLASS or len(counts) != 53:
raise SystemExit(f"wrong composition: {counts}")
args.output.parent.mkdir(parents=True, exist_ok=True)
with args.output.open("w", encoding="utf-8") as f:
for row in rows:
f.write(json.dumps(row, ensure_ascii=False, sort_keys=True) + "\n")
manifest = {
"dataset_name": "en-or-not OOD set",
"created": str(date.today()),
"source": {
"name": "Tatoeba sentence export",
"url": SOURCE_URL,
"release": SOURCE_RELEASE,
"sha256": actual_sha256,
"license": "CC-BY 2.0",
},
"selection": {
"seed": "sha256(text)",
"total_examples": len(rows),
"english_examples": counts["en"],
"non_english_examples": len(rows) - counts["en"],
"non_english_languages": sorted(LANGUAGES),
"language_counts": dict(sorted(counts.items())),
"max_utf8_bytes": 256,
"filters": ["3-276 Unicode characters", "3-256 UTF-8 bytes", "contains a letter", "normalized whitespace"],
},
"exclusion_boundary": {
"documented_training_sources": [
"Project Gutenberg collections",
"papluca/language-identification",
"oscar-corpus/oscar (optional)",
],
"guarantee": "source-level exclusion only; zero lexical overlap with undisclosed training artifacts is not guaranteed",
},
}
args.manifest.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({"rows": len(rows), "languages": len(counts), "sha256": actual_sha256}, indent=2))
if __name__ == "__main__":
main()
|