iolai26-solve / solver /preprocess.py
rvpant
final submission
a1bdb40
Raw
History Blame Contribute Delete
15.1 kB
"""Parse Linguini puzzles: normalize text, parse the context (pipe tables,
numbered/lettered lists, pairs) and the query into answerable items."""
from __future__ import annotations
import re
import unicodedata
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
_ITEM_PREFIX = re.compile(r"^\s*\(?(\d{1,3})[\.\)]\s+")
_LETTER_PREFIX = re.compile(r"^\s*\(?([A-Z])[\.\)]\s+")
_BLANK_MARK = re.compile(r"\((\d{1,3})\)")
_BLANK_LINE = re.compile(r"_{2,}|…|\.{4,}")
# non-pipe two-side separators, tried in order on non-table lines
_SEPARATORS = [
("tab", re.compile(r"\t+")),
("equals", re.compile(r"\s+=\s+")),
("emdash", re.compile(r"\s+—\s+")),
("endash", re.compile(r"\s+–\s+")),
("arrow", re.compile(r"\s*(?:->|→)\s*")),
("hyphen", re.compile(r"\s+-\s+")),
("means", re.compile(r"\s+means\s+", re.IGNORECASE)),
]
# common work-language names (queries say "Translate into English:")
_WORK_LANG_NAMES = {
"eng": "english", "fra": "french", "spa": "spanish", "por": "portuguese",
"rus": "russian", "deu": "german",
}
def normalize(text: str) -> str:
"""NFC-normalize, unify exotic whitespace/quotes. Keeps diacritics, tone
marks, case, and punctuation (EM comparison is punctuation-sensitive)."""
if text is None:
return ""
t = unicodedata.normalize("NFC", str(text))
t = t.replace(" ", " ")
t = re.sub(r"[ \t]+", " ", t)
return t.strip()
def tokenize(s: str) -> List[str]:
"""Unicode word tokenization. Keeps combining marks, word-internal
apostrophes/hyphens, and subscript/superscript markers (tone letters,
person markers like you_{sg})."""
s = normalize(s)
return re.findall(r"[^\s,;.!?()\[\]\"«»|]+", s)
def strip_punct(tok: str) -> str:
return tok.strip(",;.!?()[]\"«»").strip()
@dataclass
class Pair:
src: str # task-language side by convention
tgt: str # work-language side (gloss/translation)
sep: str = ""
line_no: int = -1
label: str = "" # numbered prefix if the line carried one
@dataclass
class QueryItem:
number: str # label as it appeared ("17", "3", "") — "" for bare lines
text: str
direction: Optional[str] = None # "to_task" | "to_work" | None
has_blank: bool = False
row: Optional[List[str]] = None # for table-blank items: full row cells
blank_col: Optional[int] = None # which cell holds this item's (k) marker
@dataclass
class Puzzle:
id: str
context: str
query: str
work_lang: str = ""
task_lang: str = ""
task_type: str = ""
eval_type: str = ""
pairs: List[Pair] = field(default_factory=list)
items: List[QueryItem] = field(default_factory=list)
hints: List[str] = field(default_factory=list)
tables: List[List[List[str]]] = field(default_factory=list) # blocks of rows of cells
numbered: Dict[str, str] = field(default_factory=dict) # "1" -> form (list contexts)
lettered: Dict[str, str] = field(default_factory=dict) # "A" -> meaning
def _split_cells(line: str) -> List[str]:
return [c.strip() for c in line.split("|")]
def _strip_item_prefix(line: str) -> Tuple[str, str]:
"""Returns (label, rest). Label may be a number or capital letter."""
m = _ITEM_PREFIX.match(line)
if m:
return m.group(1), line[m.end():].strip()
m = _LETTER_PREFIX.match(line)
if m:
return m.group(1), line[m.end():].strip()
return "", line.strip()
def _looks_header(cells: List[str]) -> bool:
"""A table header names languages/columns: 'Proto-Chamic | Tsat | meaning'."""
if len(cells) < 2:
return False
tail = cells[-1].lower()
if tail in ("meaning", "meanings", "translation", "translations", "english",
"value", "values", "gloss"):
return True
# all cells capitalized single-ish words with no digits — likely names
ok = 0
for c in cells:
if c and not any(ch.isdigit() for ch in c) and c[0].isupper() and len(c.split()) <= 3:
ok += 1
return ok == len(cells) and len(cells) >= 3
def parse_context(ctx: str) -> Tuple[List[Pair], List[str], List[List[List[str]]], Dict[str, str], Dict[str, str]]:
"""Parse context into (pairs, hints, tables, numbered, lettered)."""
pairs: List[Pair] = []
hints: List[str] = []
tables: List[List[List[str]]] = []
numbered: Dict[str, str] = {}
lettered: Dict[str, str] = {}
cur_table: List[List[str]] = []
for i, raw in enumerate(str(ctx).splitlines()):
line = normalize(raw)
if not line:
if cur_table:
tables.append(cur_table)
cur_table = []
continue
label, body = _strip_item_prefix(line)
if "|" in body:
cells = _split_cells(body)
if _looks_header(cells) and not cur_table:
hints.append(line)
continue
cur_table.append(cells)
has_blank = bool(_BLANK_MARK.search(body))
if len(cells) >= 2 and cells[0] and cells[-1] and not has_blank:
pairs.append(Pair(src=cells[0], tgt=cells[-1], sep="pipe",
line_no=i, label=label))
if label and not has_blank:
numbered[label] = cells[0]
continue
if cur_table:
tables.append(cur_table)
cur_table = []
# non-pipe separators (= , — , tab ...)
matched = False
for name, rx in _SEPARATORS:
parts = rx.split(body, maxsplit=1)
if len(parts) == 2 and parts[0].strip() and parts[1].strip():
pairs.append(Pair(src=parts[0].strip(), tgt=parts[1].strip(),
sep=name, line_no=i, label=label))
if label:
# the full line is the referable entry ("equalities (1-9)")
numbered[label] = body
matched = True
break
if matched:
continue
# single-column list entries (match_letters forms/meanings)
if label:
if label.isdigit():
numbered[label] = body
else:
lettered[label] = body
continue
hints.append(line)
if cur_table:
tables.append(cur_table)
return pairs, hints, tables, numbered, lettered
_INSTRUCTION_VERBS = (
r"(translate|fill|write|spell|determine|give|complete|convert|match|answer|"
r"say|pair|transcribe|provide|express|render|decipher|find|identify|"
r"choose|select|here|below|these|the following)"
)
_INSTRUCTION_RX = re.compile(r"^" + _INSTRUCTION_VERBS + r"\b", re.IGNORECASE)
_INSTRUCTION_ANY_RX = re.compile(r"\b" + _INSTRUCTION_VERBS + r"\b", re.IGNORECASE)
def _is_instruction(line: str) -> bool:
"""Instruction lines are work-language imperatives ("Translate into X:").
Matching is verb-anchored — a bare trailing colon is NOT enough, because
task-language forms can end in ':' (length marks: "si teŋku bugdiŋi:").
A line that ends with ':' AND contains an instruction verb anywhere is
also an instruction ("In Drehu tusi is 'book'. Translate from Drehu:")."""
line = (line or "").strip()
if not line:
return False
if _BLANK_MARK.search(line) or "|" in line:
return False
if _INSTRUCTION_RX.match(line):
return True
return line.endswith(":") and bool(_INSTRUCTION_ANY_RX.search(line))
def parse_query(query: str) -> Tuple[List[QueryItem], List[str]]:
"""Split query into answerable items + instruction lines.
Item sources, in the order encountered:
- (k)-markers inside lines (usually pipe rows): one item per marker, with
the row cells and blank column recorded;
- numbered lines "17. ..." (numbering may continue the context's);
- bare non-instruction lines: one item per line.
"""
text = str(query or "")
items: List[QueryItem] = []
instructions: List[str] = []
_TERMINAL = (".", "!", "?", ":", ";", '"', "”", "’")
for raw in text.splitlines():
line = normalize(raw)
if not line:
continue
marks = _BLANK_MARK.findall(line)
if marks:
cells = _split_cells(line) if "|" in line else [line]
for k in marks:
blank_col = next(
(ci for ci, c in enumerate(cells) if f"({k})" in c), None)
items.append(QueryItem(
number=k, text=line, has_blank=True,
row=cells if len(cells) > 1 else None, blank_col=blank_col))
continue
if "|" in line:
label, body = _strip_item_prefix(line)
if label:
# numbered table row = one item; the answer fills whichever
# column the context table has that this row lacks
items.append(QueryItem(number=label, text=body,
row=_split_cells(body)))
else:
instructions.append(line) # header/echo row
continue
label, body = _strip_item_prefix(line)
if label:
items.append(QueryItem(number=label, text=body,
has_blank=bool(_BLANK_LINE.search(body))))
continue
if _is_instruction(line):
instructions.append(line)
continue
if items and items[-1].number and not items[-1].text.rstrip().endswith(_TERMINAL):
items[-1].text += " " + line # wrapped continuation of a numbered item
continue
items.append(QueryItem(number="", text=line,
has_blank=bool(_BLANK_LINE.search(line))))
# when the query has numbered items, stray unnumbered lines around them
# are notes ("spoken on Bvuŋkaden"), not answerable items
if any(it.number for it in items):
items = [it for it in items if it.number]
# items with numeric labels answer in label order when labels are complete
if items and all(it.number.isdigit() for it in items):
items.sort(key=lambda it: int(it.number))
return items, instructions
def detect_direction(item_text: str, task_material: str, work_material: str,
instructions: List[str], work_lang: str) -> str:
"""Per-item direction: does the answer belong to the task language
('to_task') or the work language ('to_work')?
1. Explicit instruction: "into English" (work-lang name) vs "into X".
2. Script similarity: if the item text overlaps the task-language material
character-wise, it is task-language text needing analysis (to_work).
"""
joined = (" ".join(instructions) + " " + item_text).lower()
wl_name = _WORK_LANG_NAMES.get(work_lang.split("_")[0][:3].lower(), "")
m = re.search(r"(?:into|in|to)\s+(?:the\s+)?([A-Za-zÀ-ž’' -]{2,30}?)\s*(?:language)?\s*[:.]", joined + ":")
if m:
named = m.group(1).strip().lower()
if wl_name and wl_name in named:
return "to_work"
if named and not any(w in named for w in ("digit", "numeral", "number", "blank")):
return "to_task"
sim_task = _char_overlap(item_text, task_material)
sim_work = _char_overlap(item_text, work_material)
return "to_work" if sim_task >= sim_work else "to_task"
def _char_overlap(s: str, material: str, n: int = 3) -> float:
s_ = "".join(s.lower().split())
m_ = "".join(material.lower().split())
if len(s_) < n or len(m_) < n:
return 0.0
grams = {s_[i : i + n] for i in range(len(s_) - n + 1)}
hits = sum(1 for g in grams if g in m_)
return hits / len(grams)
_RANGE_RX = re.compile(r"\((\d{1,3})\s*[–—-]\s*(\d{1,3})\)")
def _items_from_context(p: Puzzle) -> List[QueryItem]:
"""When the query is instruction-only ("Fill in the blanks (1–14)",
"Determine the correct correspondences", "Write the equalities (1–9) in
numerals"), the answerable items live in the CONTEXT: (k) blank markers,
or the numbered list entries. Last resort: the query itself is one item."""
rng = _RANGE_RX.search(p.query or "")
lo, hi = (int(rng.group(1)), int(rng.group(2))) if rng else (None, None)
def in_range(k: str) -> bool:
return lo is None or (k.isdigit() and lo <= int(k) <= hi)
ctx_blanks: List[QueryItem] = []
for raw in str(p.context).splitlines():
line = normalize(raw)
for k in _BLANK_MARK.findall(line):
if not in_range(k):
continue
cells = _split_cells(line) if "|" in line else [line]
blank_col = next((ci for ci, c in enumerate(cells) if f"({k})" in c), None)
ctx_blanks.append(QueryItem(number=k, text=line, has_blank=True,
row=cells if len(cells) > 1 else None,
blank_col=blank_col))
if ctx_blanks:
ctx_blanks.sort(key=lambda it: int(it.number))
return ctx_blanks
if p.numbered and (p.task_type == "match_letters" or rng or p.lettered):
keys = sorted((k for k in p.numbered if in_range(k)), key=int)
if keys:
return [QueryItem(number=k, text=p.numbered[k]) for k in keys]
q = normalize(p.query)
return [QueryItem(number="", text=q)] if q else []
def parse_puzzle(row: dict) -> Puzzle:
"""Build a Puzzle from a CSV/dataset row (id, context, query, work_lang,
task_lang, task_type, eval_type)."""
ctx = str(row.get("context", "") or "")
p = Puzzle(
id=str(row.get("id", "")),
context=ctx,
query=str(row.get("query", "") or ""),
work_lang=str(row.get("work_lang", "") or ""),
task_lang=str(row.get("task_lang", "") or ""),
task_type=str(row.get("task_type", "") or "").strip().lower(),
eval_type=str(row.get("eval_type", "") or ""),
)
p.pairs, p.hints, p.tables, p.numbered, p.lettered = parse_context(ctx)
p.items, instructions = parse_query(p.query)
p.hints.extend(instructions)
# letter-labelled query entries are answer OPTIONS when digit-labelled
# items coexist (match tasks list both: "19. form ... S. meaning")
digit_items = [it for it in p.items if it.number.isdigit()]
letter_items = [it for it in p.items if it.number and not it.number.isdigit()]
if digit_items and letter_items:
for it in letter_items:
p.lettered[it.number] = it.text
p.items = digit_items
if not p.items:
p.items = _items_from_context(p)
task_material = " ".join(x.src for x in p.pairs) + " " + " ".join(p.numbered.values())
work_material = " ".join(x.tgt for x in p.pairs) + " " + " ".join(p.lettered.values())
for it in p.items:
if it.row is not None:
continue # table-blank items get direction from their row in the router
it.direction = detect_direction(it.text, task_material, work_material,
instructions, p.work_lang)
return p