"""Parse Linguini puzzles: normalize text, parse the context (pipe tables, numbered/lettered lists, pairs) and the query into answerable items.""" from __future__ import annotations import re import unicodedata from dataclasses import dataclass, field from typing import Dict, List, Optional, Tuple _ITEM_PREFIX = re.compile(r"^\s*\(?(\d{1,3})[\.\)]\s+") _LETTER_PREFIX = re.compile(r"^\s*\(?([A-Z])[\.\)]\s+") _BLANK_MARK = re.compile(r"\((\d{1,3})\)") _BLANK_LINE = re.compile(r"_{2,}|…|\.{4,}") # non-pipe two-side separators, tried in order on non-table lines _SEPARATORS = [ ("tab", re.compile(r"\t+")), ("equals", re.compile(r"\s+=\s+")), ("emdash", re.compile(r"\s+—\s+")), ("endash", re.compile(r"\s+–\s+")), ("arrow", re.compile(r"\s*(?:->|→)\s*")), ("hyphen", re.compile(r"\s+-\s+")), ("means", re.compile(r"\s+means\s+", re.IGNORECASE)), ] # common work-language names (queries say "Translate into English:") _WORK_LANG_NAMES = { "eng": "english", "fra": "french", "spa": "spanish", "por": "portuguese", "rus": "russian", "deu": "german", } def normalize(text: str) -> str: """NFC-normalize, unify exotic whitespace/quotes. Keeps diacritics, tone marks, case, and punctuation (EM comparison is punctuation-sensitive).""" if text is None: return "" t = unicodedata.normalize("NFC", str(text)) t = t.replace(" ", " ") t = re.sub(r"[ \t]+", " ", t) return t.strip() def tokenize(s: str) -> List[str]: """Unicode word tokenization. Keeps combining marks, word-internal apostrophes/hyphens, and subscript/superscript markers (tone letters, person markers like you_{sg}).""" s = normalize(s) return re.findall(r"[^\s,;.!?()\[\]\"«»|]+", s) def strip_punct(tok: str) -> str: return tok.strip(",;.!?()[]\"«»").strip() @dataclass class Pair: src: str # task-language side by convention tgt: str # work-language side (gloss/translation) sep: str = "" line_no: int = -1 label: str = "" # numbered prefix if the line carried one @dataclass class QueryItem: number: str # label as it appeared ("17", "3", "") — "" for bare lines text: str direction: Optional[str] = None # "to_task" | "to_work" | None has_blank: bool = False row: Optional[List[str]] = None # for table-blank items: full row cells blank_col: Optional[int] = None # which cell holds this item's (k) marker @dataclass class Puzzle: id: str context: str query: str work_lang: str = "" task_lang: str = "" task_type: str = "" eval_type: str = "" pairs: List[Pair] = field(default_factory=list) items: List[QueryItem] = field(default_factory=list) hints: List[str] = field(default_factory=list) tables: List[List[List[str]]] = field(default_factory=list) # blocks of rows of cells numbered: Dict[str, str] = field(default_factory=dict) # "1" -> form (list contexts) lettered: Dict[str, str] = field(default_factory=dict) # "A" -> meaning def _split_cells(line: str) -> List[str]: return [c.strip() for c in line.split("|")] def _strip_item_prefix(line: str) -> Tuple[str, str]: """Returns (label, rest). Label may be a number or capital letter.""" m = _ITEM_PREFIX.match(line) if m: return m.group(1), line[m.end():].strip() m = _LETTER_PREFIX.match(line) if m: return m.group(1), line[m.end():].strip() return "", line.strip() def _looks_header(cells: List[str]) -> bool: """A table header names languages/columns: 'Proto-Chamic | Tsat | meaning'.""" if len(cells) < 2: return False tail = cells[-1].lower() if tail in ("meaning", "meanings", "translation", "translations", "english", "value", "values", "gloss"): return True # all cells capitalized single-ish words with no digits — likely names ok = 0 for c in cells: if c and not any(ch.isdigit() for ch in c) and c[0].isupper() and len(c.split()) <= 3: ok += 1 return ok == len(cells) and len(cells) >= 3 def parse_context(ctx: str) -> Tuple[List[Pair], List[str], List[List[List[str]]], Dict[str, str], Dict[str, str]]: """Parse context into (pairs, hints, tables, numbered, lettered).""" pairs: List[Pair] = [] hints: List[str] = [] tables: List[List[List[str]]] = [] numbered: Dict[str, str] = {} lettered: Dict[str, str] = {} cur_table: List[List[str]] = [] for i, raw in enumerate(str(ctx).splitlines()): line = normalize(raw) if not line: if cur_table: tables.append(cur_table) cur_table = [] continue label, body = _strip_item_prefix(line) if "|" in body: cells = _split_cells(body) if _looks_header(cells) and not cur_table: hints.append(line) continue cur_table.append(cells) has_blank = bool(_BLANK_MARK.search(body)) if len(cells) >= 2 and cells[0] and cells[-1] and not has_blank: pairs.append(Pair(src=cells[0], tgt=cells[-1], sep="pipe", line_no=i, label=label)) if label and not has_blank: numbered[label] = cells[0] continue if cur_table: tables.append(cur_table) cur_table = [] # non-pipe separators (= , — , tab ...) matched = False for name, rx in _SEPARATORS: parts = rx.split(body, maxsplit=1) if len(parts) == 2 and parts[0].strip() and parts[1].strip(): pairs.append(Pair(src=parts[0].strip(), tgt=parts[1].strip(), sep=name, line_no=i, label=label)) if label: # the full line is the referable entry ("equalities (1-9)") numbered[label] = body matched = True break if matched: continue # single-column list entries (match_letters forms/meanings) if label: if label.isdigit(): numbered[label] = body else: lettered[label] = body continue hints.append(line) if cur_table: tables.append(cur_table) return pairs, hints, tables, numbered, lettered _INSTRUCTION_VERBS = ( r"(translate|fill|write|spell|determine|give|complete|convert|match|answer|" r"say|pair|transcribe|provide|express|render|decipher|find|identify|" r"choose|select|here|below|these|the following)" ) _INSTRUCTION_RX = re.compile(r"^" + _INSTRUCTION_VERBS + r"\b", re.IGNORECASE) _INSTRUCTION_ANY_RX = re.compile(r"\b" + _INSTRUCTION_VERBS + r"\b", re.IGNORECASE) def _is_instruction(line: str) -> bool: """Instruction lines are work-language imperatives ("Translate into X:"). Matching is verb-anchored — a bare trailing colon is NOT enough, because task-language forms can end in ':' (length marks: "si teŋku bugdiŋi:"). A line that ends with ':' AND contains an instruction verb anywhere is also an instruction ("In Drehu tusi is 'book'. Translate from Drehu:").""" line = (line or "").strip() if not line: return False if _BLANK_MARK.search(line) or "|" in line: return False if _INSTRUCTION_RX.match(line): return True return line.endswith(":") and bool(_INSTRUCTION_ANY_RX.search(line)) def parse_query(query: str) -> Tuple[List[QueryItem], List[str]]: """Split query into answerable items + instruction lines. Item sources, in the order encountered: - (k)-markers inside lines (usually pipe rows): one item per marker, with the row cells and blank column recorded; - numbered lines "17. ..." (numbering may continue the context's); - bare non-instruction lines: one item per line. """ text = str(query or "") items: List[QueryItem] = [] instructions: List[str] = [] _TERMINAL = (".", "!", "?", ":", ";", '"', "”", "’") for raw in text.splitlines(): line = normalize(raw) if not line: continue marks = _BLANK_MARK.findall(line) if marks: cells = _split_cells(line) if "|" in line else [line] for k in marks: blank_col = next( (ci for ci, c in enumerate(cells) if f"({k})" in c), None) items.append(QueryItem( number=k, text=line, has_blank=True, row=cells if len(cells) > 1 else None, blank_col=blank_col)) continue if "|" in line: label, body = _strip_item_prefix(line) if label: # numbered table row = one item; the answer fills whichever # column the context table has that this row lacks items.append(QueryItem(number=label, text=body, row=_split_cells(body))) else: instructions.append(line) # header/echo row continue label, body = _strip_item_prefix(line) if label: items.append(QueryItem(number=label, text=body, has_blank=bool(_BLANK_LINE.search(body)))) continue if _is_instruction(line): instructions.append(line) continue if items and items[-1].number and not items[-1].text.rstrip().endswith(_TERMINAL): items[-1].text += " " + line # wrapped continuation of a numbered item continue items.append(QueryItem(number="", text=line, has_blank=bool(_BLANK_LINE.search(line)))) # when the query has numbered items, stray unnumbered lines around them # are notes ("spoken on Bvuŋkaden"), not answerable items if any(it.number for it in items): items = [it for it in items if it.number] # items with numeric labels answer in label order when labels are complete if items and all(it.number.isdigit() for it in items): items.sort(key=lambda it: int(it.number)) return items, instructions def detect_direction(item_text: str, task_material: str, work_material: str, instructions: List[str], work_lang: str) -> str: """Per-item direction: does the answer belong to the task language ('to_task') or the work language ('to_work')? 1. Explicit instruction: "into English" (work-lang name) vs "into X". 2. Script similarity: if the item text overlaps the task-language material character-wise, it is task-language text needing analysis (to_work). """ joined = (" ".join(instructions) + " " + item_text).lower() wl_name = _WORK_LANG_NAMES.get(work_lang.split("_")[0][:3].lower(), "") m = re.search(r"(?:into|in|to)\s+(?:the\s+)?([A-Za-zÀ-ž’' -]{2,30}?)\s*(?:language)?\s*[:.]", joined + ":") if m: named = m.group(1).strip().lower() if wl_name and wl_name in named: return "to_work" if named and not any(w in named for w in ("digit", "numeral", "number", "blank")): return "to_task" sim_task = _char_overlap(item_text, task_material) sim_work = _char_overlap(item_text, work_material) return "to_work" if sim_task >= sim_work else "to_task" def _char_overlap(s: str, material: str, n: int = 3) -> float: s_ = "".join(s.lower().split()) m_ = "".join(material.lower().split()) if len(s_) < n or len(m_) < n: return 0.0 grams = {s_[i : i + n] for i in range(len(s_) - n + 1)} hits = sum(1 for g in grams if g in m_) return hits / len(grams) _RANGE_RX = re.compile(r"\((\d{1,3})\s*[–—-]\s*(\d{1,3})\)") def _items_from_context(p: Puzzle) -> List[QueryItem]: """When the query is instruction-only ("Fill in the blanks (1–14)", "Determine the correct correspondences", "Write the equalities (1–9) in numerals"), the answerable items live in the CONTEXT: (k) blank markers, or the numbered list entries. Last resort: the query itself is one item.""" rng = _RANGE_RX.search(p.query or "") lo, hi = (int(rng.group(1)), int(rng.group(2))) if rng else (None, None) def in_range(k: str) -> bool: return lo is None or (k.isdigit() and lo <= int(k) <= hi) ctx_blanks: List[QueryItem] = [] for raw in str(p.context).splitlines(): line = normalize(raw) for k in _BLANK_MARK.findall(line): if not in_range(k): continue cells = _split_cells(line) if "|" in line else [line] blank_col = next((ci for ci, c in enumerate(cells) if f"({k})" in c), None) ctx_blanks.append(QueryItem(number=k, text=line, has_blank=True, row=cells if len(cells) > 1 else None, blank_col=blank_col)) if ctx_blanks: ctx_blanks.sort(key=lambda it: int(it.number)) return ctx_blanks if p.numbered and (p.task_type == "match_letters" or rng or p.lettered): keys = sorted((k for k in p.numbered if in_range(k)), key=int) if keys: return [QueryItem(number=k, text=p.numbered[k]) for k in keys] q = normalize(p.query) return [QueryItem(number="", text=q)] if q else [] def parse_puzzle(row: dict) -> Puzzle: """Build a Puzzle from a CSV/dataset row (id, context, query, work_lang, task_lang, task_type, eval_type).""" ctx = str(row.get("context", "") or "") p = Puzzle( id=str(row.get("id", "")), context=ctx, query=str(row.get("query", "") or ""), work_lang=str(row.get("work_lang", "") or ""), task_lang=str(row.get("task_lang", "") or ""), task_type=str(row.get("task_type", "") or "").strip().lower(), eval_type=str(row.get("eval_type", "") or ""), ) p.pairs, p.hints, p.tables, p.numbered, p.lettered = parse_context(ctx) p.items, instructions = parse_query(p.query) p.hints.extend(instructions) # letter-labelled query entries are answer OPTIONS when digit-labelled # items coexist (match tasks list both: "19. form ... S. meaning") digit_items = [it for it in p.items if it.number.isdigit()] letter_items = [it for it in p.items if it.number and not it.number.isdigit()] if digit_items and letter_items: for it in letter_items: p.lettered[it.number] = it.text p.items = digit_items if not p.items: p.items = _items_from_context(p) task_material = " ".join(x.src for x in p.pairs) + " " + " ".join(p.numbered.values()) work_material = " ".join(x.tgt for x in p.pairs) + " " + " ".join(p.lettered.values()) for it in p.items: if it.row is not None: continue # table-blank items get direction from their row in the router it.direction = detect_direction(it.text, task_material, work_material, instructions, p.work_lang) return p