"""Shared MorphyNet morpheme-tokenizer core (v0.1). Single source of truth, imported by build/analysis scripts (DRY + reproducible). Keys / V_A = supplement (function words) + MorphyNet inflectional forms + known_vocab (incl. derivational words AS ATOMIC). Derivational *splits* are deferred (#2); derivational words still count as recognizable (otherwise -ly adverbs like `happily`/`quickly`, which are derivation-only, would become ). """ import csv, os, re, sys from collections import Counter from pathlib import Path MIN_RESIDUAL = 3 KEY_CUTOFF = 160_000 # operating point: use the top-160k MorphyNet keys by wordfreq (raw morphynet files untouched) OOV_SUFFIXES = [("ing", "V"), ("ies", "N"), ("es", "N"), ("ed", "V"), ("s", "N")] # longest-first; 's via contraction() WORD_RE = re.compile(r"[^\W\d_]+(?:['-][^\W\d_]+)*|\d+") # Unicode letters (keep accents) + internal '/-; OR a number CLEAN_MORPH = re.compile(r"^[a-z]+$") PRON_WH_S = {"he", "she", "that", "what", "there", "who", "where"} # + 's -> base is NT_IRREGULAR = {"won't": ["will", "n't"], "can't": ["can", "n't"], "shan't": ["shall", "n't"], "ain't": ["ain't"]} SPECIAL = {"let's": ["let", "us"]} def contraction(word): if "'" not in word: return None if word in SPECIAL: return SPECIAL[word] if word in ("it's", "why's", "how's"): # ambiguous is/has -> keep whole return [word] if word.endswith("'d"): # all 'd ambiguous had/would -> whole return [word] if word.endswith("n't"): # Keep `n't` as a CLITIC morpheme, not the free word `not` -- same treatment as the `'s` clitic below. # Expanding to `not` scrambles word order under subject-aux inversion (`don't they -> do not they`, # a string that never occurs in English), and the model learns from the linear order. `n't` is a # distinct bound allomorph of negation, so it earns its own morpheme. (`won't -> will n't` etc. keep # the recovered base via NT_IRREGULAR, avoiding PTB's `wo n't`/`ca n't`.) if word in NT_IRREGULAR: return NT_IRREGULAR[word] b = word[:-3] return [b, "n't"] if b else None for cl, exp in (("'re", "are"), ("'ve", "have"), ("'ll", "will")): if word.endswith(cl): b = word[:-3] return [b, exp] if b else None if word.endswith("'m"): b = word[:-2] return [b, "am"] if b else None if word.endswith("'s"): b = word[:-2] if not b: return None return [b, "is"] if b in PRON_WH_S else [b, "'s"] # he's->he is ; dog's->dog 's if word.endswith("s'") and len(word) > 2: # PLURAL POSSESSIVE: `skateboards'` -> `skateboards` + `'s`. The trailing `'` is the genitive on a # plural noun; drop it, keep the `s`, emit the same `'s` clitic as the singular. The base flows # through analyze() (via the base-recursion above), so `skateboards' -> skate board s 's`. Restricted # to `s'` so it catches plurals (dogs', teachers') and s-ending singulars (James') but NOT g-dropping # elisions, which end in other letters (goin', nothin', ol'). return [word[:-1], "'s"] return None # --------------------------------------------------------------------------------------------------------- # THE LOADED MORPHEME TABLES. resources/morphemes/ holds EXACTLY the files the tokenizer reads -- nothing # else lives there. Everything under resources/ proper is a work-in-progress, a generator input, or raw # detector output, and is NOT loaded. If you can see it in morphemes/, it is live. # # Loaded IN ORDER; a LATER file overrides an earlier one, and every conflict is logged loudly. # Override the whole list with $BABYLM_SUPPLEMENTS (colon-separated) or the supplement_path argument. # # NEVER PUT A `_proposed` / `_candidates` FILE IN HERE. Those are raw detector output. Only files a human has # signed off belong in morphemes/. `phantom_fixes_proposed.tsv` (625 auto-repairs; the file is now called # base_forms_round2_proposed.tsv, and is still NOT loaded) was loaded from # 2026-07-11 to 2026-07-12 BY MISTAKE -- and, worse, the hand-reviewed phantom_verified.tsv was NOT loaded at # the time, so the machine's guesses were live while the human's decisions sat on the shelf. guard_output() # now refuses to let any script write into this list. # --------------------------------------------------------------------------------------------------------- MORPHEMES = "resources/morphemes" DEFAULT_SUPPLEMENTS = ( # DERIVATION loads FIRST, on purpose: it is the most GENERIC source (MorphyNet's rows, gated by the # researcher's handmade vocab), so every hand-curated file below overrides it. Load it LAST instead and a # single stray row silently beats a deliberate decision -- during the merge it briefly did exactly that, # turning `anyone` into `any one`, `dryer` into `dry er`, `potter` into `pot er`. f"{MORPHEMES}/derivation_verified.tsv", # 2,804 derivational splits (tokenizer-todos #3). Each row was # approved by resources/handmade_vocab.tsv and then hand-curated. f"{MORPHEMES}/derivation_verified2.tsv", # 7,618 more: the frequency-ranked shortlist of MorphyNet rows # the handmade vocab had no verdict on, audited in two rounds # (95% usable) and then hand-corrected. Loads with round 1, and # is likewise overridden by every hand-curated file below. f"{MORPHEMES}/additional_words_agents.tsv", # 541 agent-generated inflectional/derivational splits # (`heard -> hear ed`, `words -> word s`), reviewed by the # researcher. Loads EARLY (generic backfill) so every specific # hand-curated file below still overrides it on any conflict. f"{MORPHEMES}/morphynet_gap_fills_quite_sure.tsv", # 1332 DECOMPOSED gap-fills from full MorphyNet # (babylm/eval words OOV at top-160k but recognized at full # vocab), HAND-REVIEWED by a 14-subagent pass (110 fixed, 95 made # atomic). Generic backfill -> loads EARLY, overridable. f"{MORPHEMES}/morph_supplement.tsv", f"{MORPHEMES}/demonyms.tsv", # place + "an" demonyms the suffix rule mis-split or missed: # mallorcan -> mallorca an, croatian -> croatia an. The base # place then decomposes further (croat ia an) -- intended, to # maximize sharing. f"{MORPHEMES}/names.tsv", f"{MORPHEMES}/names_male.tsv", f"{MORPHEMES}/names_female.tsv", f"{MORPHEMES}/places.tsv", f"{MORPHEMES}/name_top55_verified.tsv", # 47 verified names + phœbe->phoebe, urler->ursula. # Its 6 corpus ARTIFACTS (cou, ö, nköö, i'ii, alphahff, iím) are # deliberately left as COMMENTS so they can never be loaded. f"{MORPHEMES}/superlatives.tsv", f"{MORPHEMES}/non-english.tsv", f"{MORPHEMES}/run_together.tsv", f"{MORPHEMES}/elision_verified.tsv", f"{MORPHEMES}/syllable_hyphen_verified.tsv", f"{MORPHEMES}/hyphen_keep_whole.tsv", # hyphenated words kept whole / normalized, overriding the # hyphen-split rule. Only the lexicalized/idiomatic/proper ones # are here; every other hyphenated word splits on its hyphens. f"{MORPHEMES}/phantom_verified.tsv", # phantom stems REVIEWED BY HAND from corpus passages (#7) f"{MORPHEMES}/base_forms_verified.tsv", # BASE forms whose inflected form was already covered but which # were themselves atomic or : `besiegers -> besiege er s` # was handled while `besieger` was . Proposed by # find_missing_base_forms.py, then reviewed by hand. Loads AFTER # phantom_verified because it builds on those repairs. f"{MORPHEMES}/ghost_stem_fixes.tsv", # 261 rows. MorphyNet's inflection table lemmatises with an # ORTHOGRAPHIC GHOST -- an Early Modern spelling that occurs in # our Gutenberg text, so the phantom detector's `stem in CORPUS` # test called it a real word and never looked: `seemed -> # seeme ed` (1,892 tokens), `waiting -> waite ing` (1,100), # `helped -> helpe ed` (740). Presence cannot refute a phantom; # only presence in PROPORTION can (seeme 1x vs seemed 1,892x). f"{MORPHEMES}/atomic_corrections.tsv", # 433 INFLECTIONAL splits whose stem is not a word: # `number -> numb er`, `seemed -> seeme ed` (an Early Modern # spelling that occurs in our Gutenberg text, which is exactly # why the phantom detector had been blind to it). Proposed by # propose_infl_atomic.py, then reviewed by 28 agents over every # row, each finding re-checked by a second agent told to refute # it. Fixes 91% of the inflection bug mass. f"{MORPHEMES}/archaic_verbs.tsv", # Early Modern verb agreement from the Gutenberg slice, which # MorphyNet does not cover at all: `cometh -> come eth`, # `hast -> have st`. Finishes a convention already in use # (`hath -> have th`, `shalt -> shall t`). 385 corpus tokens # were . Names in -eth (kenneth, gwyneth) excluded by # cap-ratio; the silent e is restored (`ride eth`, not `rid eth`). f"{MORPHEMES}/more_fixes.tsv", # ZERO-CHANGE IRREGULARS. The past tense of `shed` IS `shed`, # but MorphyNet writes the segmentation `shed|ed` -- a suffix # that is not in the string. We key INFL by FORM, so EVERY # `shed` (the garden shed included) came out as `shed ed`, and # it corrupted `toolshed -> tool shed` into `tool shed ed` via # recursive expansion. 176 corpus tokens. See §11 of # docs/why-morphynet-needs-curation.md. The ROOT fix belongs in # load_resources(): reject any row where lemma == form and the # segmentation still splits. This file is the interim patch. f"{MORPHEMES}/nonce.tsv", # wug / fep / blicket / dax / wampimuk. ATOMIC BY DESIGN, and # loaded precisely BECAUSE they never occur in the corpus: the # wug test measures generalisation to an unseen word, so if the # tokenizer shatters `wug` into or letter-pieces the # benchmark stops measuring what it is for. (blicket/fep/wug are # 40% of the whole eval-vs-corpus gap by token count -- comps # repeats them 55,584 times each.) f"{MORPHEMES}/posessives.tsv", # `beverly's -> beverly 's`. The BLiMP vocabulary gap is mostly # this: 20+ of its 78 out-of-corpus words are a first name + # `'s`. Needs a general rule eventually -- these are hand-listed. f"{MORPHEMES}/linker_exceptions.tsv", # words kept WHOLE because dropping the LINKER would collide: # `woodsy` would become `wood y` = woody. 6 measured collisions # + the system family. Loads LATE so it overrides derivation. f"{MORPHEMES}/inflections.tsv", # 259 regular inflections MorphyNet's INFL table missed, so the # tokenizer kept them whole (`girls`, `allowed`, `covered`). The # SAFE backfill for under-split Finding 1 -- explicit entries, not # a blanket known-word peel (which would mis-split news/loss). # Built by scripts/build_inflections_tsv.py from the audit. f"{MORPHEMES}/compounds.tsv", # transparent compounds/prefixed words left atomic that should # split: `birthday -> birth day`, `impossible -> in possible`, # `preschool -> pre school`. Under-split Finding 3. f"{MORPHEMES}/goal1_derivations.tsv", # productive, POS-gated MorphyNet derivations + `-men` plurals # (`happiness -> happy ness`, `teacher -> teach er`, `horsemen -> # horse man s`), hand-reviewed. From the overnight morpheme- # reduction probe (scripts/goal1_reduction_probe.py). ~-200 morphemes. f"{MORPHEMES}/british-to-american.tsv", # British->American spelling normalization, ISOLATED here so it # can be toggled on/off by including/excluding this one file. # RHS is the AMERICAN spelling as a WHOLE WORD (`analogue -> # analog`, `equaliser -> equalizer`); _expand_supplement # re-analyzes it, so `equalizer` chains to `equal ize er` and # `analog` resolves to the corpus atom. NO other tsv may bake in # normalization (no `discolour -> dis color`). See backlog #7. # Loads LATE so the respelling overrides any split of the British # form. Only 2 rows now; populate over time. f"{MORPHEMES}/mojibake-fixes.tsv", # the `í` (U+00ED) corruption: contraction mojibake -> apostrophe # form (`itís -> it's`), and accented names + letter-i mojibake -> # ASCII (`rodríguez -> rodriguez`, `hís -> his`). Built by # scripts/build_mojibake_fixes.py. See Goal-3 findings. f"{MORPHEMES}/typos.tsv", f"{MORPHEMES}/men_compounds.tsv", # `-men`/`-man` compounds: keep `men` whole (X men, the suppletive # plural), NOT `X man s`. Singulars are `X man`. See §-men in # docs/tokenizer-design-decisions and men_compounds header. f"{MORPHEMES}/oov_rare_words_approved.tsv", # REVIEWED subset of proposals/oov_rare_words.tsv -- only the # hand-decided `word -> morphemes` entries, verified against # babylm+eval usage. See docs/oov-rare-words-approved.md. f"{MORPHEMES}/ambiguous.tsv", # words that CANNOT be split because the split is genuinely # ambiguous: `leaves` = leaf+s (noun) OR leave+s (verb), so forcing # either reading throws away information -> keep atomic. Bare-word # format (word alone = atomic). Loads LAST so "keep whole" overrides # any earlier split (e.g. INFL's `leaves -> leave s`). SUPPLEMENT # also beats INFL in analyze(), so the atomic entry always wins. ) def guard_output(path): """Refuse to write a HUMAN-OWNED file. Call before EVERY write under resources/. Two invariants, both learned by breaking them: 1. NEVER write a `*_verified.tsv`. That suffix means a human signed it off. A script that regenerates one silently destroys hours of hand review -- and the researcher cannot tell, because the file is still there and still looks plausible. 2. NEVER write a file that is in DEFAULT_SUPPLEMENTS. Same reason: it is live in the tokenizer. LAYOUT + NAMING CONTRACT enforced here: resources/morphemes/*.tsv LOADED. Human-owned. NEVER machine-written. resources/*_candidates.tsv machine-written, never loaded, safe to regenerate resources/*_proposed.tsv machine-written, never loaded, safe to regenerate *_verified.tsv human-signed-off; never machine-written, wherever it lives Returns the path so it can be used inline: `with open(mt.guard_output(OUT), "w") as f:` """ p = Path(path) if p.name.endswith("_verified.tsv"): raise RuntimeError( f"REFUSING to write {p.name}: the `_verified` suffix means a human signed it off.\n" f"Write to a `_proposed` / `_candidates` name instead; promote it to `_verified` only by hand." ) if any(Path(s).name == p.name for s in DEFAULT_SUPPLEMENTS): raise RuntimeError( f"REFUSING to write {p.name}: it is LOADED by the tokenizer (DEFAULT_SUPPLEMENTS).\n" f"Write to a `_proposed` / `_candidates` name instead." ) return p def _supplement_paths(repo, supplement_path): """Resolve which supplement TSVs to load: explicit arg > $BABYLM_SUPPLEMENTS (colon-separated) > default.""" if supplement_path is not None: paths = [supplement_path] if isinstance(supplement_path, (str, Path)) else list(supplement_path) else: env = os.environ.get("BABYLM_SUPPLEMENTS") paths = env.split(":") if env else list(DEFAULT_SUPPLEMENTS) out = [] for p in paths: p = Path(p) out.append(p if p.is_absolute() else repo / p) return out def load_resources(repo, supplement_path=None): # Shipped fast path: fully-resolved dict (no wordfreq dependency, no TSV parsing). _resolved = Path(repo) / "resources_resolved.json" if _resolved.exists(): import json as _json _r = _json.loads(_resolved.read_text(encoding="utf-8")) return {_k: set(_v) if isinstance(_v, list) else _v for _k, _v in _r.items()} """Load supplement TSV(s) + MorphyNet inflection/known. supplement_path may be one path or a list. Multiple files are merged in order; a LATER file overrides an earlier one, and any conflicting key is reported loudly (never silently resolved). FILE FORMAT — fields are whitespace-separated (a tab is conventional, spaces also work): word morph1 morph2 ... -> `word` analyses to those morphemes. e.g. `on-ly only` word -> ATOMIC: `word` analyses to itself. word word -> IDENTICAL to the bare form above; both give val == [word]. So the two atomic conventions are interchangeable — use whichever you prefer. CAVEAT: the KEY is lowercased, the VALUE is not. Since the corpus is lowercased at tokenization, an uppercase value (`sarah Sarah`) would create a morpheme that matches nothing. Keep values lowercase. `#` starts a comment; everything after it on the line is ignored. """ repo = Path(repo) supplement, origin, conflicts = {}, {}, [] for supp_path in _supplement_paths(repo, supplement_path): for line in open(supp_path, encoding="utf-8"): line = line.split("#", 1)[0].strip() if not line: continue p = line.split() key = p[0].lower() val = p[1:] if len(p) > 1 else [key] if key in supplement and supplement[key] != val: conflicts.append((key, origin[key], supplement[key], supp_path.name, val)) supplement[key] = val origin[key] = supp_path.name infl, known = {}, set() # A form can carry SEVERAL rows with DIFFERENT lemmas -- `helped` is offered as both `help|ed` and the # Early Modern `helpe|ed`. The old code did infl.setdefault(form, ...), i.e. FIRST ROW IN FILE ORDER wins, # which is an accident: `helpe` precedes `help`, so `helped -> helpe ed` and the correct row was lost. # Collect every candidate, then pick the one whose LEMMA is the most frequent ENGLISH word (wordfreq). # wordfreq, not corpus count: corpus count picks `well` over `good` for `better` and so drifts with the # training data; wordfreq is external and stable. First-seen is the tie-break, so the 987k forms with no # competitor are unaffected. Only ~988 forms change. See section 8 of docs/why-morphynet-needs-curation.md. infl_cands: dict[str, list] = {} for r in csv.reader(open(repo / "data" / "morphynet" / "eng.inflectional.v1.tsv", encoding="utf-8"), delimiter="\t"): if len(r) < 4: continue lemma, form, seg = r[0].lower(), r[1].lower(), r[3].lower() known.add(lemma); known.add(form) if seg == "-": continue pieces = seg.split("|") if all(CLEAN_MORPH.match(p) for p in pieces): infl_cands.setdefault(form, []).append((lemma, pieces)) for form, cands in infl_cands.items(): if len(cands) == 1 or len({tuple(p) for _, p in cands}) == 1: infl[form] = cands[0][1] # no genuine disagreement -> nothing to rank else: infl[form] = max(cands, key=lambda lp: _wordfreq(lp[0]))[1] for r in csv.reader(open(repo / "data" / "morphynet" / "eng.derivational.v1.tsv", encoding="utf-8"), delimiter="\t"): if len(r) >= 6: known.add(r[0].lower()); known.add(r[1].lower()) res = {"SUPPLEMENT": supplement, "INFL": infl, "known": known} _report_conflicts(conflicts, res) return res def _report_conflicts(conflicts, res): """Warn only about conflicts that actually CHANGE THE OUTPUT. Supplement values are RECURSIVELY EXPANDED (see _expand_supplement), so two files can write the same answer two different ways and be identical in effect: morph_supplement.tsv caretaking care taking -> expands to ['care', 'take', 'ing'] phantom_verified.tsv caretaking care take ing -> expands to ['care', 'take', 'ing'] Comparing the RAW values calls that a conflict and sends the researcher hunting for a problem that does not exist. Compare what the tokenizer actually emits instead, and stay quiet when it is the same. """ real = [] for k, f1, v1, f2, v2 in conflicts: if _expand_supplement(k, v1, res, 0) != _expand_supplement(k, v2, res, 0): real.append((k, f1, v1, f2, v2)) benign = len(conflicts) - len(real) if real: print(f"WARNING: {len(real)} conflicting supplement key(s) that CHANGE THE OUTPUT; later file wins:", file=sys.stderr) for k, f1, v1, f2, v2 in real[:20]: print(f" {k}: {f1}={v1} -> {f2}={v2}", file=sys.stderr) if benign: print(f"({benign} further duplicate key(s) write the same answer a different way — no effect.)", file=sys.stderr) def rank_keys(known): """Rank known_vocab keys by general-English frequency (wordfreq), most-frequent first.""" import importlib as _il; word_frequency = _il.import_module("wordfreq").word_frequency # hidden from HF check_imports; never executed (resolved-JSON fast path) return sorted(known, key=lambda w: (-word_frequency(w, "en"), w)) def truncate_to_topk(res, top_k=KEY_CUTOFF, ranked=None): """Restrict known_vocab + INFL to the top_k most-frequent keys. top_k=None keeps all. Raw morphynet files untouched. Pass a precomputed `ranked` (from rank_keys) to avoid re-ranking across repeated calls (e.g. the coverage curve). """ if ranked is None: ranked = rank_keys(res["known"]) kept = set(ranked) if top_k is None else set(ranked[:top_k]) return {"SUPPLEMENT": res["SUPPLEMENT"], "INFL": {f: p for f, p in res["INFL"].items() if f in kept}, "known": kept} def peel_oov(word): for suf, cls in OOV_SUFFIXES: if word.endswith(suf) and len(word) - len(suf) >= MIN_RESIDUAL: return word[: -len(suf)], suf, cls return None # --- inflectional stem restoration (NOT derivation) ------------------------------------------------- # Peeling `-ies`/`-ed`/`-ing` off an OOV word usually leaves a non-word (`opportunit`, `carv`, `runn`). # Restore the real stem orthographically. Suffixes are normalised to reusable morphemes (ies/es -> s). SUFFIX_MORPH = {"ies": "s", "es": "s", "s": "s", "ed": "ed", "ing": "ing"} def _stem_candidates(stem, suf): """Orthographic ways the surface stem could map back to a real word.""" if suf in ("ing", "ed"): c = [stem, stem + "e"] # walk / carve if len(stem) >= 3 and stem[-1] == stem[-2] and stem[-1] not in "aeiou": c.append(stem[:-1]) # runn -> run (consonant doubling) return c if suf == "ies": return [stem + "y", stem + "ie"] # opportunit -> opportunity ; mov -> movie if suf == "es": return [stem, stem + "e"] # box ; witness return [stem] _WF = None # injectable for tests; lazily bound to wordfreq.word_frequency def _wordfreq(w): global _WF if _WF is None: import importlib as _il; word_frequency = _il.import_module("wordfreq").word_frequency # hidden from HF check_imports; never executed (resolved-JSON fast path) _WF = word_frequency return _WF(w, "en") def restore_stem(stem, suf, res): """Return the unique real stem, or None if ZERO or MORE THAN ONE candidate is real. Ambiguity => None => the word is left alone (principle 3). `hoping` has both `hop` and `hope` as real words, so it is deliberately NOT peeled. wordfreq (not the dictionary) decides what is 'real', because MorphyNet's known-vocab contains phantoms like `walke`/`talke` that would otherwise win. """ cands = [c for c in dict.fromkeys(_stem_candidates(stem, suf)) if c in res["known"]] strong = [c for c in cands if _wordfreq(c) >= 1e-6] if len(strong) == 1: return strong[0] if len(strong) > 1: return None # genuinely ambiguous -> keep atomic return cands[0] if len(cands) == 1 else None # rare-but-real stem, unique MAX_SUPP_DEPTH = 5 # guard 3: bounds cycles (a->b, b->a) in supplement values UNEXPANDED = {} # supplement values that could not be fully analysed -> reported loudly, never silent def _expand_supplement(word, pieces, res, _depth): """Expand a supplement value through analyze(), so values may be written as WORDS, not pre-split morphemes. Guards: (1) self-reference (`sarah sarah`) never recurses; (2) a piece that cannot be analysed is kept literal AND recorded in UNEXPANDED rather than degrading into ; (3) depth cap bounds cycles. """ if _depth >= MAX_SUPP_DEPTH: return list(pieces) out = [] for p in pieces: if p == word: # guard 1 out.append(p) continue sub, _ = analyze(p, res, _depth + 1) if any(x.startswith(" ['2','0','2','6'] (digit-split; revisitable) if word in res["SUPPLEMENT"]: return _expand_supplement(word, res["SUPPLEMENT"][word], res, _depth), "supplement" c = contraction(word) if c is not None: # contraction() is a pure string function with no access to `res`, so it hands back the BASE as a # raw string (`skateboard's` -> ['skateboard', "'s"]). Re-analyze that base so it decomposes like it # would anywhere else -> ['skate', 'board', "'s"]. Recurse ONLY the base (c[0]); the clitic tail # (`'s`, `is`, `n't`, `are`, ...) is already a final morpheme and must be kept verbatim -- analysing # `'s` on its own yields . Guard c[0] != word so the kept-whole cases (`it's`, `i'd`), which # return a single-element [word], never recurse into themselves. if len(c) > 1 and c[0] != word: base, _ = analyze(c[0], res, _depth + 1) return base + c[1:], "contraction" return c, "contraction" if word in res["INFL"]: return res["INFL"][word], "inflection" if "-" in word: # Split a hyphenated word when every part is itself covered -> reusable, meaningful, unambiguous. # This runs BEFORE the `known` check ON PURPOSE: MorphyNet's dictionary catalogs thousands of # hyphenated compounds as whole words (cable-car, non-stick, well-known), so with `known` first, # whether a compound split depended on the ACCIDENT of MorphyNet listing it. The lexicalized ones # that must stay whole (band-aid, hip-hop, blue-collar) are held atomic by hyphen_keep_whole.tsv / # syllable_hyphen_verified.tsv, which are SUPPLEMENTS and so already win above this point. parts = [p for p in word.split("-") if p] if len(parts) >= 2: out = [] for part in parts: sub, _ = analyze(part, res, _depth + 1) if any(x.startswith(" opportunity + s if restored: sub, _ = analyze(restored, res, _depth + 1) return sub + [SUFFIX_MORPH.get(suf, suf)], "inflection" return [f"", suf], "oov_affix" return [f""], "atomic_unk" # --------------------------------------------------------------------------- # CORPUS PREPROCESSING (a DATA concern, deliberately NOT part of analyze()). # Emulates what the training-ingest script will eventually do. Set BABYLM_RAW=1 to disable and read raw lines. # --------------------------------------------------------------------------- CURLY = str.maketrans({"’": "'", "‘": "'", "ʼ": "'"}) # typographic apostrophes -> ASCII HEADER_RE = re.compile(r"^= = = .* = = =\s*$") # BabyLM document separator: `= = = childes/.../x.cha = = =` TIER_RE = re.compile(r"^%[a-z]+:") # CHILDES dependent tiers: %int:, %add: (annotation, not speech) SPK_RE = re.compile(r"^\*[A-Za-z]{2,5}:[ \t]*") # CHILDES speaker tag: strip prefix, KEEP the utterance def clean_line(line): """Return the cleaned line, or None if the line should be skipped entirely. - skip `= = = ... = = =` document headers and `%tier:` annotation lines (pure metadata) - strip the leading `*SPK:` speaker tag but keep the utterance (the speech is on that line) - normalise typographic apostrophes to ASCII `'` so contractions survive tokenization """ if HEADER_RE.match(line) or TIER_RE.match(line): return None return SPK_RE.sub("", line, count=1).translate(CURLY) def _lines(fp): raw = os.environ.get("BABYLM_RAW") == "1" for line in open(fp, encoding="utf-8"): if raw: yield line continue cleaned = clean_line(line) if cleaned is not None: yield cleaned def count_corpus(corpus_dir): """Return (CORPUS_FREQ Counter, {domain: Counter}).""" cf, df = Counter(), {} for fp in sorted(Path(corpus_dir).glob("*.train.txt")): c = Counter() for line in _lines(fp): c.update(WORD_RE.findall(line.lower())) df[fp.name.split(".")[0]] = c cf.update(c) return cf, df CASED_WORD_RE = re.compile(r"[^\W\d_]+") # case-PRESERVING; WORD_RE is used on lowercased text def count_corpus_cased(corpus_dir): """Return (capitalised Counter, lowercase Counter), both keyed by the LOWERCASED word. A proper name stays capitalised mid-sentence, so a high cap-ratio is the cheapest reliable name signal. Needed because count_corpus() lowercases and thus destroys it. """ cap, low = Counter(), Counter() for fp in sorted(Path(corpus_dir).glob("*.train.txt")): for line in _lines(fp): for w in CASED_WORD_RE.findall(line): (cap if w[0].isupper() else low)[w.lower()] += 1 return cap, low def is_probable_name(word, cap, low, min_tokens=20, ratio=0.75): """True if `word` is capitalised at least `ratio` of the time -- i.e. it is a proper name. Guards the morphological repair scripts, which otherwise happily produce `holmes -> holm s` (Sherlock), `venus -> venue s`, `torres -> tor s`. """ n = cap[word] + low[word] return n >= min_tokens and cap[word] / n >= ratio def stream_corpus(corpus_dir): """Yield corpus words in order (for chunk-level analysis).""" for fp in sorted(Path(corpus_dir).glob("*.train.txt")): for line in _lines(fp): for w in WORD_RE.findall(line.lower()): yield w