any2human / app /engine /lexical.py
idnameraj's picture
Block marginal WordNet senses that broke second-pass wording.
cfb5d84
Raw
History Blame Contribute Delete
53.6 kB
"""Conservative, context-aware vocabulary refinement (no hard-coded word lists)."""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass, field
from functools import lru_cache
from typing import Any
from lemminflect import getInflection
from wordfreq import zipf_frequency
from app.config import (
ENGINE_LEXICAL_MAX_FREQUENCY_GAP,
ENGINE_LEXICAL_MAX_HARDER_GAP,
ENGINE_LEXICAL_MAX_CHANGES,
ENGINE_LEXICAL_MAX_SIMPLER_GAP,
ENGINE_LEXICAL_MIN_ZIPF,
ENGINE_LEXICAL_PREFER_SIMPLER,
ENGINE_WORDNET_LEXICON,
)
from app.engine.models import LexicalChange
from app.engine.quality import substitution_pos_stable
from app.pipeline.nlp import get_nlp
_POS_MAP = {"NOUN": "n", "VERB": "v", "ADJ": "a", "ADV": "r"}
_PROTECTED_MARKER = re.compile(r"ZZPROTECTED(?:URL|EMAIL|PATH)\d+ZZ", re.I)
_WORD = re.compile(r"[A-Za-z][A-Za-z'-]*")
_CITATION = re.compile(
r"(?:\[[0-9,\s-]+\]|\([A-Z][^()]{0,60},\s*(?:19|20)\d{2}\))"
)
_QUOTES = frozenset({'"', "“", "”", "‘", "’"})
_CLEFT_PREFIX = re.compile(r"^it is\b", re.I)
logger = logging.getLogger("plainrewrite.lexical")
@dataclass
class LexicalResult:
text: str
changes: list[LexicalChange] = field(default_factory=list)
confidence: float = 0.0
reason: str = ""
@lru_cache(maxsize=1)
def _get_wordnet() -> Any | None:
try:
import wn
wn.config.allow_multithreading = True
return wn.Wordnet(ENGINE_WORDNET_LEXICON)
except Exception as exc:
logger.warning("Open English WordNet unavailable: %s", exc)
return None
def lexical_resource_available() -> bool:
return _get_wordnet() is not None
def _context_terms(doc, target) -> set[str]:
terms: set[str] = set()
for token in doc:
if token.i == target.i or token.is_stop or not token.is_alpha:
continue
if token.pos_ not in _POS_MAP and token.pos_ != "PROPN":
continue
for value in (token.text.lower(), token.lemma_.lower()):
if len(value) >= 3:
terms.add(value)
return terms
def _terms_from_text(text: str, stop_words: set[str]) -> set[str]:
terms = {match.group(0).lower() for match in _WORD.finditer(text or "")}
return {term for term in terms if len(term) >= 3 and term not in stop_words}
def _gloss_terms(synset, stop_words: set[str]) -> tuple[set[str], set[str]]:
definition = _terms_from_text(synset.definition(), stop_words)
examples: set[str] = set()
try:
for example in synset.examples():
examples.update(_terms_from_text(example, stop_words))
except Exception:
pass
return definition, examples
def _sense_score(synset, context: set[str], stop_words: set[str]) -> float:
if not context:
return 0.0
definition, examples = _gloss_terms(synset, stop_words)
definition_overlap = context & definition
if not definition_overlap:
return 0.0
example_overlap = context & examples
weighted = len(definition_overlap) + (0.25 * len(example_overlap))
return min(1.0, weighted / max(1, min(5, len(context))))
def _eligible(token, doc) -> bool:
if token.pos_ not in _POS_MAP:
return False
lemma = token.lemma_.lower()
if (
token.pos_ == "PROPN"
or token.ent_type_
or token.is_stop
or not token.is_alpha
or len(lemma) < 3
or token.dep_ in {"aux", "auxpass", "neg", "mark"}
or _PROTECTED_MARKER.search(token.text)
):
return False
if token.tag_ == "VBG" and token.dep_ in {"ROOT", "csubj", "nsubj", "attr"}:
return False
if token.tag_ == "VBG" and token.i <= 2 and _CLEFT_PREFIX.search(doc.text):
return False
if lemma in {"be", "have", "do"}:
return False
if _CLEFT_PREFIX.search(doc.text) and token.text.lower() in {"it"}:
return False
# Freeze "* to …" constructions (equipped to, bring to, get to, …).
if token.i + 1 < len(doc) and doc[token.i + 1].lower_ == "to":
return False
# Freeze passive verbal heads ("are brought by …").
if token.pos_ == "VERB" and (
token.dep_ in {"auxpass"}
or any(child.dep_ == "auxpass" for child in token.children)
):
return False
return True
def _impact_rank(token) -> int:
if token.pos_ == "ADV":
return 0
if token.pos_ == "ADJ":
return 1
if token.pos_ == "VERB" and token.dep_ != "ROOT":
return 2
if token.pos_ == "VERB":
return 3
if token.pos_ == "NOUN" and token.dep_ in {
"attr",
"acomp",
"oprd",
"pobj",
"dobj",
"appos",
}:
return 4
if token.pos_ == "NOUN":
return 5
return 6
def _inflect(lemma: str, token) -> str | None:
forms = getInflection(lemma, tag=token.tag_)
value = forms[0] if forms else lemma
if not value or not _WORD.fullmatch(value) or " " in value or "_" in value:
return None
if token.text.isupper():
return value.upper()
if token.text[:1].isupper():
return value[:1].upper() + value[1:]
return value.lower()
def _neighbor_words(doc, token) -> tuple[str, str]:
left = ""
right = ""
if token.i > 0:
prev = doc[token.i - 1]
if prev.is_alpha:
left = prev.text.lower()
if token.i + 1 < len(doc):
nxt = doc[token.i + 1]
if nxt.is_alpha:
right = nxt.text.lower()
return left, right
def _object_lemma(token) -> str:
for child in token.children:
if child.dep_ in {"dobj", "obj"}:
return child.lemma_.lower()
return ""
def _phrase_zipf(word: str, left: str, right: str) -> float:
scores = [zipf_frequency(word, "en")]
if left:
scores.append(zipf_frequency(f"{left} {word}", "en"))
if right:
scores.append(zipf_frequency(f"{word} {right}", "en"))
return max(scores)
def _disaster_vo_blocked(
source_lemma: str,
candidate_lemma: str,
token,
*,
left: str,
right: str,
) -> bool:
"""Hard meaning-break blockers used even in aggressive classical mode."""
if source_lemma == candidate_lemma:
return False
obj = _object_lemma(token)
src_v = zipf_frequency(source_lemma, "en")
cand_v = zipf_frequency(candidate_lemma, "en")
if token.pos_ == "VERB" and obj:
src_vo = zipf_frequency(f"{source_lemma} {obj}", "en")
cand_vo = zipf_frequency(f"{candidate_lemma} {obj}", "en")
src_spec = src_vo - src_v
cand_spec = cand_vo - cand_v
# resolve→decide / establish→launch: commoner verb inflates VO, sense drifts.
if src_vo >= 3.8 and cand_v - src_v >= 0.35 and cand_spec <= src_spec + 0.08:
return True
if src_vo >= 3.8 and cand_vo >= src_vo and cand_spec + 0.08 < src_spec:
return True
# Crowded VO neighborhood: block only when specificity gets worse
# (decide/determine/influence issues), allow near-parity (resolve→settle).
if src_vo >= 4.0 and cand_vo >= 4.0 and cand_spec + 0.05 < src_spec:
return True
# Absolute floor: weak-specificity VO targets are almost always sense drift.
if obj and cand_spec < -0.08 and src_vo >= 4.0:
return True
# deliver→function/serve style on service objects.
if cand_spec < -0.30 and src_spec < -0.15 and cand_spec < src_spec + 0.40:
return True
if token.pos_ == "VERB" and right:
src_bi = zipf_frequency(f"{source_lemma} {right}", "en")
cand_bi = zipf_frequency(f"{candidate_lemma} {right}", "en")
# respond promptly → answer promptly (ungrammatical with following "to").
if (
right == "promptly"
and cand_v >= 5.0
and cand_v - src_v >= 0.40
and cand_bi + 0.10 >= src_bi
):
return True
return False
def _collocation_ok(
source_lemma: str,
candidate_lemma: str,
token,
*,
left: str,
right: str,
classical_strict: bool = False,
headword_upgrade: bool = False,
classical_aggressive: bool = False,
) -> bool:
"""Reject swaps that collapse local collocations (algorithmic, no denylist)."""
# Aggressive mode: restore synonym throughput, keep only disaster VO brakes.
if classical_aggressive and classical_strict and not headword_upgrade:
if _disaster_vo_blocked(
source_lemma, candidate_lemma, token, left=left, right=right
):
return False
classical_strict = False
obj = _object_lemma(token)
src_v = zipf_frequency(source_lemma, "en")
cand_v = zipf_frequency(candidate_lemma, "en")
checks: list[tuple[float, float]] = []
if obj:
checks.append(
(
zipf_frequency(f"{source_lemma} {obj}", "en"),
zipf_frequency(f"{candidate_lemma} {obj}", "en"),
)
)
if right:
checks.append(
(
zipf_frequency(f"{source_lemma} {right}", "en"),
zipf_frequency(f"{candidate_lemma} {right}", "en"),
)
)
if left:
checks.append(
(
zipf_frequency(f"{left} {source_lemma}", "en"),
zipf_frequency(f"{left} {candidate_lemma}", "en"),
)
)
for source_score, candidate_score in checks:
slack = 0.35 if classical_strict else 0.55
if source_score >= 2.2 and candidate_score + slack < source_score:
return False
if source_score >= 3.0 and candidate_score < (2.4 if classical_strict else 2.0):
return False
if source_score >= 3.5 and candidate_score + (
0.45 if classical_strict else 0.85
) < source_score:
return False
if source_score >= 4.0 and candidate_score < 2.0:
return False
if classical_strict and source_score >= 2.5:
# Specificity = bigram − bare verb. Blocks resolve→decide / respond→answer
# where a more common verb inflates raw zipf but weakens the pair.
src_spec = source_score - src_v
cand_spec = candidate_score - cand_v
if headword_upgrade:
# maintain→keep: allow a specificity drop only with a clear everyday leap.
if cand_spec + 0.05 < src_spec and cand_v - src_v < 0.90:
return False
else:
if cand_spec + 0.02 < src_spec:
return False
if candidate_score > source_score and cand_spec < src_spec:
return False
if cand_v >= 5.15 and cand_v - src_v >= 0.45:
return False
# Verb + object: prefer attested collocations; block free WordNet drift.
if token.pos_ == "VERB" and obj:
source_obj = zipf_frequency(f"{source_lemma} {obj}", "en")
candidate_obj = zipf_frequency(f"{candidate_lemma} {obj}", "en")
src_spec = source_obj - src_v
cand_spec = candidate_obj - cand_v
if classical_strict:
if headword_upgrade:
if candidate_obj + 0.05 < source_obj or candidate_obj < 2.0:
return False
if cand_spec + 0.05 < src_spec and cand_v - src_v < 0.90:
return False
return True
if cand_spec + 0.02 < src_spec:
return False
if candidate_obj > source_obj and cand_spec < src_spec:
return False
if cand_v >= 5.15 and cand_v - src_v >= 0.45:
return False
if cand_spec < -0.30 and src_spec < -0.20 and cand_spec < src_spec + 0.55:
return False
# Both rare with this object → synonym roulette (decide/define issues).
if source_obj < 2.2 and candidate_obj < 2.2:
return False
if source_obj >= 2.0 and candidate_obj + 0.20 < source_obj:
return False
if candidate_obj < 2.0:
return False
if abs(cand_v - src_v) < 0.40 and cand_spec <= src_spec + 0.05:
return False
return True
if source_obj >= 1.8 and candidate_obj + 0.35 < source_obj:
return False
if source_obj < 1.5 and candidate_obj < 1.5:
# Both rare with this object — require the candidate verb itself to be
# clearly more everyday, handled elsewhere; still block near-ties.
return candidate_obj + 0.1 >= source_obj
elif classical_strict and token.pos_ == "VERB":
# No object: block leaps into ultra-common verbs unless a true headword leap.
if cand_v >= 5.15 and cand_v - src_v >= 0.45:
if not headword_upgrade or cand_v - src_v < 0.90:
return False
return True
def _peer_cycle_blocked(
source_lemma: str,
candidate_lemma: str,
source_frequency: float,
candidate_frequency: float,
*,
classical_strict: bool,
classical_aggressive: bool = False,
allow_headword_upgrade: bool = False,
) -> bool:
"""Block synonym oscillation.
Strict mode blocks near-peer swaps. Aggressive mode only blocks clear
demotions (rarer candidate) so each pass can still diverge.
"""
if not classical_strict or allow_headword_upgrade:
return False
if classical_aggressive:
# Allow near-peer motion across passes (keep↔maintain); only block
# severe demotions into rare wording.
return candidate_frequency + 0.85 < source_frequency
if candidate_frequency + 0.20 < source_frequency:
return True
if (
abs(candidate_frequency - source_frequency) < 0.40
and candidate_frequency < source_frequency + 0.45
):
return True
return False
def _frequency_ok(
source_frequency: float,
candidate_frequency: float,
*,
classical_aggressive: bool = False,
) -> bool:
if classical_aggressive:
# Allow broader everyday band so multi-pass wording can keep moving.
if candidate_frequency < max(3.2, ENGINE_LEXICAL_MIN_ZIPF - 0.8):
return False
return abs(candidate_frequency - source_frequency) <= 1.8
if candidate_frequency < ENGINE_LEXICAL_MIN_ZIPF:
return False
if ENGINE_LEXICAL_PREFER_SIMPLER:
if candidate_frequency > source_frequency:
return (
candidate_frequency - source_frequency
<= ENGINE_LEXICAL_MAX_SIMPLER_GAP
)
return source_frequency - candidate_frequency <= ENGINE_LEXICAL_MAX_HARDER_GAP
return (
abs(candidate_frequency - source_frequency)
<= ENGINE_LEXICAL_MAX_FREQUENCY_GAP
)
def _candidate_for_synset(
synset,
token,
doc,
*,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> tuple[str, float] | None:
source = token.lemma_.lower()
source_surface = token.text.lower()
source_frequency = max(
zipf_frequency(source, "en"),
zipf_frequency(source_surface, "en"),
)
left, right = _neighbor_words(doc, token)
source_phrase = _phrase_zipf(source_surface, left, right)
try:
words = synset.words()
except Exception:
return None
ranked: list[tuple[float, float, float, str]] = []
for word in words:
lemma = (word.lemma() or "").replace("_", " ").strip().lower()
if (
not lemma
or lemma == source
or " " in lemma
or not _WORD.fullmatch(lemma)
):
continue
if not _collocation_ok(
source,
lemma,
token,
left=left,
right=right,
classical_strict=classical_strict,
headword_upgrade=False,
classical_aggressive=classical_aggressive,
):
continue
replacement = _inflect(lemma, token)
if not replacement or replacement.lower() == token.text.lower():
continue
candidate_frequency = zipf_frequency(lemma, "en")
rank_frequency = zipf_frequency(replacement.lower(), "en")
if not _frequency_ok(
source_frequency,
max(candidate_frequency, rank_frequency),
classical_aggressive=classical_aggressive,
):
continue
if _peer_cycle_blocked(
source,
lemma,
source_frequency,
max(candidate_frequency, rank_frequency),
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
):
continue
# Avoid noun swaps into ultra-common generic words (goal→end).
if (
token.pos_ == "NOUN"
and max(candidate_frequency, rank_frequency) >= 5.5
and max(candidate_frequency, rank_frequency) - source_frequency >= 0.5
):
continue
phrase = _phrase_zipf(replacement.lower(), left, right)
phrase_slack = (
0.55
if classical_aggressive
else (0.25 if classical_strict else 0.55)
)
if phrase + phrase_slack < source_phrase:
continue
if left and token.pos_ == "NOUN":
source_bigram = zipf_frequency(f"{left} {source_surface}", "en")
candidate_bigram = zipf_frequency(
f"{left} {replacement.lower()}", "en"
)
if source_bigram >= 3.0 and candidate_bigram + 0.45 < source_bigram:
continue
if (
token.pos_ == "NOUN"
and left
and token.i > 0
and doc[token.i - 1].pos_ == "ADJ"
and phrase > source_phrase + 0.25
):
continue
if not substitution_pos_stable(doc.text, token.i, replacement):
continue
ranked.append(
(
max(candidate_frequency, rank_frequency) - source_frequency,
-abs(
max(candidate_frequency, rank_frequency)
- (
source_frequency
+ (0.55 if ENGINE_LEXICAL_PREFER_SIMPLER else 0.0)
)
),
phrase,
-float(len(replacement)),
replacement,
)
)
if not ranked:
return None
# Prefer a clear everyday upgrade when available (assist→help), otherwise
# the nearest modest step that still passes frequency checks.
clear = [item for item in ranked if item[0] >= 0.45]
pool = clear if clear else [item for item in ranked if item[0] >= 0.0]
if not pool:
pool = ranked
if classical_aggressive:
# Prefer surface-different wording so multi-pass rewrites keep moving.
pool.sort(
key=lambda item: (
abs(len(item[4]) - len(source_surface))
+ sum(1 for a, b in zip(item[4].lower(), source_surface) if a != b),
item[0],
item[2],
),
reverse=True,
)
else:
pool.sort(key=lambda item: (item[1], item[0], item[2], item[3]), reverse=True)
best = pool[0][4]
return best, pool[0][0]
def _article_for(word: str) -> str:
return "an" if word[:1].lower() in {"a", "e", "i", "o", "u"} else "a"
def _synset_single_word_lemmas(synset) -> list[str]:
lemmas: list[str] = []
try:
words = synset.words()
except Exception:
return lemmas
for word in words:
lemma = (word.lemma() or "").replace("_", " ").strip().lower()
if lemma and " " not in lemma and _WORD.fullmatch(lemma):
lemmas.append(lemma)
return lemmas
def _source_lemma_rank(synset, source_lemma: str) -> int:
"""Lower rank = source is a more canonical member of this synset."""
lemmas = _synset_single_word_lemmas(synset)
try:
return lemmas.index(source_lemma)
except ValueError:
return 99
def _candidate_lemma_for_replacement(synset, token, replacement: str) -> str:
"""Map an inflected replacement back to a synset lemma when possible."""
surface = replacement.lower()
for lemma in _synset_single_word_lemmas(synset):
form = _inflect(lemma, token)
if form and form.lower() == surface:
return lemma
if lemma == surface:
return lemma
return surface
def _unglossed_verb_allowed(
synset,
*,
source_lemma: str,
cand_lemma: str,
stop_words: set[str],
classical_strict: bool = False,
) -> bool:
"""Allow unglossed verb swaps only with strong sense evidence.
Blocks preserve→continue while keeping purchase→buy, assist→help,
construct→build, and require→need on canonical senses.
"""
defn_terms = _terms_from_text(synset.definition(), stop_words)
lemmas = _synset_single_word_lemmas(synset)
others = [lemma for lemma in lemmas if lemma != source_lemma]
source_rank = _source_lemma_rank(synset, source_lemma)
# Never demote a headword to a secondary lemma (keep→maintain).
if classical_strict and source_rank == 0 and cand_lemma in lemmas[1:]:
return False
# Definition names the candidate in a sense where source is near-canonical.
if (
cand_lemma in defn_terms
and source_rank <= 2
and len(lemmas) >= 3
):
return True
# Two-word everyday headword pair (buy/purchase, end/terminate).
if (
len(others) == 1
and others[0] == cand_lemma
and lemmas
and lemmas[0] == cand_lemma
and source_rank == 1
):
defn_l = (synset.definition() or "").lower()
if cand_lemma in defn_terms:
return True
# "obtain by purchase" style definitions name the source as the method.
if f"by {source_lemma}" in defn_l or f"of {source_lemma}" in defn_l:
return True
# Headword in a multi-lemma sense (construct/build/make).
if (
source_rank == 0
and len(lemmas) >= 3
and cand_lemma in lemmas[:3]
and not classical_strict
):
return True
return False
def _polish_headword_upgrade(synset, source_lemma: str, cand_lemma: str) -> bool:
"""Polish-only: allow source→headword upgrades in small verb synsets.
Example: maintain→keep on keep/maintain/hold. Blocks identify→place where
the source is already the headword.
"""
lemmas = _synset_single_word_lemmas(synset)
if not lemmas or lemmas[0] != cand_lemma:
return False
source_rank = _source_lemma_rank(synset, source_lemma)
return 1 <= source_rank <= 4 and 2 <= len(lemmas) <= 5
def _headword_candidate_for_synset(
synset,
token,
doc,
*,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> tuple[str, float] | None:
"""Try the synset headword directly for polish-only verb upgrades."""
lemmas = _synset_single_word_lemmas(synset)
if not lemmas:
return None
head = lemmas[0]
replacement = _inflect(head, token)
if not replacement or replacement.lower() == token.text.lower():
return None
source = token.lemma_.lower()
source_surface = token.text.lower()
source_frequency = max(
zipf_frequency(source, "en"),
zipf_frequency(source_surface, "en"),
)
left, right = _neighbor_words(doc, token)
source_phrase = _phrase_zipf(source_surface, left, right)
if not _collocation_ok(
source,
head,
token,
left=left,
right=right,
classical_strict=classical_strict,
headword_upgrade=True,
classical_aggressive=classical_aggressive,
):
return None
candidate_frequency = zipf_frequency(head, "en")
rank_frequency = zipf_frequency(replacement.lower(), "en")
if not _frequency_ok(
source_frequency,
max(candidate_frequency, rank_frequency),
classical_aggressive=classical_aggressive,
):
return None
phrase = _phrase_zipf(replacement.lower(), left, right)
phrase_slack = 0.55 if classical_aggressive else (0.25 if classical_strict else 0.55)
if phrase + phrase_slack < source_phrase:
return None
if not substitution_pos_stable(doc.text, token.i, replacement):
return None
return replacement, max(candidate_frequency, rank_frequency) - source_frequency
def _pick_verb_candidate(
token,
doc,
synsets: list[Any],
context: set[str],
stop_words: set[str],
*,
effective_min: float,
gain_floor: float,
polish: bool = False,
aggressive: bool = False,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> tuple[float, str, float, Any] | None:
"""Choose a verb synonym from a few top senses with collocational safety."""
obj = _object_lemma(token)
src_lemma = token.lemma_.lower()
weak_gloss = max(float(effective_min), 0.25)
# Prenominal/predicative participles often act like adjectives
# ("satisfying customer experience") — never trust unglossed verb senses.
adjectival_participle = token.tag_ == "VBG" and token.dep_ in {
"amod",
"acomp",
"oprd",
}
options: list[tuple[float, int, float, float, str, Any]] = []
# WordNet orders senses by frequency. Without a meaning gate, a wide window
# lets marginal senses supply peers (settle "reside" → locate), so classical
# mode only considers the dominant readings.
sense_window = 3 if classical_strict else 6
ordered = sorted(
synsets[:sense_window],
key=lambda synset: (
_source_lemma_rank(synset, src_lemma),
-_sense_score(synset, context, stop_words),
),
)
for synset in ordered:
score = _sense_score(synset, context, stop_words)
picked = _candidate_for_synset(
synset,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
headword_polish = False
if (
polish
and not aggressive
and score < weak_gloss
and _source_lemma_rank(synset, src_lemma) >= 1
):
headword_pick = _headword_candidate_for_synset(
synset,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if headword_pick is not None:
picked = headword_pick
headword_polish = True
if not picked:
continue
replacement, simplicity_gain = picked
if (
score < effective_min
and simplicity_gain < gain_floor
and not headword_polish
):
continue
cand_lemma = _candidate_lemma_for_replacement(synset, token, replacement)
high_gain_ok = False
if score < weak_gloss:
if adjectival_participle:
continue
unglossed_ok = _unglossed_verb_allowed(
synset,
source_lemma=src_lemma,
cand_lemma=cand_lemma,
stop_words=stop_words,
classical_strict=classical_strict and not classical_aggressive,
)
# Polish may upgrade to the synset headword (maintain→keep) when
# the source is a secondary lemma. Aggressive/ensure stay stricter.
high_gain_ok = (
polish
and simplicity_gain >= 0.08
and _polish_headword_upgrade(synset, src_lemma, cand_lemma)
)
if classical_strict and high_gain_ok:
# Headword upgrades still need object attestation when present.
if obj:
src_c = zipf_frequency(f"{src_lemma} {obj}", "en")
cand_c = zipf_frequency(f"{cand_lemma} {obj}", "en")
if cand_c + 0.05 < src_c or cand_c < 2.0:
high_gain_ok = False
if not unglossed_ok and not high_gain_ok:
continue
multiword_parent = False
try:
for word in synset.words():
multi = (word.lemma() or "").replace("_", " ").strip().lower()
if multi.startswith(cand_lemma + " "):
multiword_parent = True
break
except Exception:
pass
if multiword_parent:
continue
src_colloc = zipf_frequency(f"{src_lemma} {obj}", "en") if obj else 0.0
cand_colloc = zipf_frequency(f"{cand_lemma} {obj}", "en") if obj else 0.0
# Strong verb+object phrases need real context gloss, not synonymy alone
# (blocks create→make results while allowing purchase→buy).
if obj and src_colloc >= 4.7 and score < weak_gloss and score < effective_min:
continue
if (
score < weak_gloss
and polish
and high_gain_ok
and obj
and (
cand_colloc + 0.05 < src_colloc
# Large collocation jumps often mark a different sense
# (identify areas → name areas), while modest ones are OK
# (maintain attitude → keep attitude).
or (src_colloc >= 4.2 and cand_colloc - src_colloc >= 0.45)
)
):
continue
if classical_strict and obj and cand_colloc < 2.0:
continue
options.append(
(
score,
_source_lemma_rank(synset, src_lemma),
simplicity_gain,
cand_colloc,
replacement,
synset,
)
)
if not options:
return None
options.sort(key=lambda item: (-item[0], item[1], -item[2], -item[3]))
score, _rank, gain, _colloc, replacement, synset = options[0]
return (max(score, effective_min), replacement, gain, synset)
def dynamic_lexical_budget(text: str, *, polish: bool = False) -> int:
"""Choose how many synonym swaps a sentence may take from its length.
Longer sentences get a larger budget so wording density scales with size
instead of a fixed environment cap.
"""
words = len(_WORD.findall(text or ""))
if words < 3:
return 0
# Normal: about one safe swap per 8 words; polish: denser (~per 4 words).
stride = 4 if polish else 8
budget = max(1, (words + stride - 1) // stride)
if polish and words >= 5:
budget = max(budget, 2)
# Soft ceiling grows with length; polish allows a bit more.
ceiling = (6 + words // 6) if polish else (4 + words // 10)
budget = min(budget, max(1, min(ceiling, 15)))
if ENGINE_LEXICAL_MAX_CHANGES > 0:
budget = min(budget, ENGINE_LEXICAL_MAX_CHANGES)
return budget
def _related_modifier_synsets(synset) -> list[Any]:
"""WordNet-linked adjective/adverb senses (no hardcoded synonym lists)."""
related: list[Any] = []
seen: set[str] = set()
for relation in ("also", "similar"):
try:
linked = list(synset.get_related(relation) or [])
except Exception:
linked = []
for item in linked:
syn_id = str(getattr(item, "id", item))
if syn_id in seen:
continue
seen.add(syn_id)
related.append(item)
if len(related) >= 12:
return related
return related
def _pick_related_modifier_candidate(
token,
doc,
synsets: list[Any],
*,
gain_floor: float,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> tuple[float, str, float, Any] | None:
"""Polish helper: try WordNet related senses when same-synset synonyms are thin."""
if classical_strict and not classical_aggressive:
return None
source_freq = max(
zipf_frequency(token.lemma_.lower(), "en"),
zipf_frequency(token.text.lower(), "en"),
)
options: list[tuple[float, float, str, Any]] = []
for synset in synsets[:2]:
for related in _related_modifier_synsets(synset):
picked = _candidate_for_synset(
related,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if not picked:
continue
replacement, simplicity_gain = picked
if simplicity_gain < gain_floor or simplicity_gain > 0.95:
continue
cand_freq = zipf_frequency(replacement.lower(), "en")
if cand_freq < ENGINE_LEXICAL_MIN_ZIPF:
continue
# Related senses are broader than true synonyms; keep them everyday
# but avoid ultra-generic leaps (effective→strong, important→big).
if cand_freq >= 5.2:
continue
if cand_freq - source_freq >= 0.85:
continue
options.append((simplicity_gain, cand_freq, replacement, related))
if not options:
return None
options.sort(key=lambda item: (item[0], item[1]), reverse=True)
gain, _freq, replacement, synset = options[0]
return (max(0.12, gain_floor), replacement, gain, synset)
def refine_sentence(
text: str,
*,
min_wsd: float,
max_changes: int | None = None,
wordnet: Any | None = None,
aggressive: bool = False,
polish: bool = False,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> LexicalResult:
"""Replace low-impact words with simpler everyday synonyms when safe."""
source = (text or "").strip()
if not source:
return LexicalResult(text=text, reason="empty")
if any(quote in source for quote in _QUOTES):
return LexicalResult(text=source, reason="quoted")
if _CITATION.search(source):
return LexicalResult(text=source, reason="citation")
if _PROTECTED_MARKER.search(source):
return LexicalResult(text=source, reason="protected")
nlp = get_nlp()
resource = wordnet if wordnet is not None else _get_wordnet()
if nlp is None or resource is None:
return LexicalResult(text=source, reason="resource_unavailable")
try:
doc = nlp(source)
except Exception:
return LexicalResult(text=source, reason="parse_failed")
tight = classical_strict and not classical_aggressive
if aggressive:
if tight:
# No forced aggressive synonym roulette without a meaning gate.
return LexicalResult(text=source, reason="classical_strict_no_aggressive")
effective_min = max(0.05, min_wsd * 0.5)
verb_gain_floor = 0.08 if classical_aggressive else 0.40
adj_adv_gain_floor = 0.0 if classical_aggressive else 0.15
elif polish:
# Polish mode should feel stronger on any sentence, not only because the
# sentence earned a larger budget. Keep the same safety rails, but allow
# slightly weaker-yet-still-supported everyday substitutions.
effective_min = max(0.05, min_wsd * 0.6) if classical_aggressive else max(0.10, min_wsd * 0.85)
verb_gain_floor = 0.08 if classical_aggressive else (0.45 if tight else 0.30)
adj_adv_gain_floor = 0.0 if classical_aggressive else (0.20 if tight else 0.10)
else:
effective_min = max(0.08, min_wsd * 0.7) if classical_aggressive else (min_wsd if not tight else max(min_wsd, 0.18))
verb_gain_floor = 0.15 if classical_aggressive else (0.45 if not tight else 0.55)
adj_adv_gain_floor = 0.0 if not tight else 0.15
stop_words = set(nlp.Defaults.stop_words)
proposals: list[tuple[float, int, float, int, int, str, Any]] = []
for token in doc:
if not _eligible(token, doc):
continue
# Prefer low-impact slots: modifiers first; nouns/verbs need stronger sense support.
context = _context_terms(doc, token)
try:
synsets = list(
resource.synsets(
token.lemma_.lower(),
pos=_POS_MAP[token.pos_],
)
)
except Exception:
continue
if not synsets:
continue
# Nouns stay on the primary sense. Verbs search canonical senses first.
if token.pos_ in {"NOUN", "VERB"}:
if token.pos_ == "VERB":
picked_verb = _pick_verb_candidate(
token,
doc,
synsets,
context,
stop_words,
effective_min=effective_min,
gain_floor=verb_gain_floor,
polish=polish,
aggressive=aggressive,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if not picked_verb:
continue
best_score, replacement, simplicity_gain, chosen = picked_verb
else:
chosen = synsets[0]
best_score = _sense_score(chosen, context, stop_words)
picked = _candidate_for_synset(
chosen,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if not picked:
continue
replacement, simplicity_gain = picked
noun_min = (
effective_min if not aggressive else max(0.08, effective_min)
)
if (
best_score < noun_min
or simplicity_gain
< (
(-0.15 if classical_aggressive else 0.25)
if polish and not aggressive
else (-0.05 if classical_aggressive else 0.35)
)
or simplicity_gain
> (1.8 if classical_aggressive else (1.45 if polish and not aggressive else 1.25))
):
continue
cand_freq = zipf_frequency(replacement.lower(), "en")
if cand_freq >= 5.5 and simplicity_gain >= (
0.7 if polish and not aggressive else 0.5
):
continue
if best_score < effective_min:
best_score = effective_min
proposals.append(
(
best_score,
_impact_rank(token),
-simplicity_gain,
token.idx,
token.i,
replacement,
chosen,
)
)
continue
else:
scored: list[tuple[float, Any]] = []
for synset in synsets:
scored.append((_sense_score(synset, context, stop_words), synset))
scored.sort(key=lambda item: item[0], reverse=True)
best_score, best_synset = scored[0]
chosen = None
if best_score >= effective_min:
chosen = best_synset
elif len(synsets) == 1:
chosen = best_synset
best_score = effective_min
elif polish and not aggressive and best_score >= effective_min * 0.75:
trial = _candidate_for_synset(
best_synset,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if trial is not None and trial[1] >= adj_adv_gain_floor:
chosen = best_synset
best_score = max(best_score, effective_min)
elif ENGINE_LEXICAL_PREFER_SIMPLER and best_score == 0.0:
trial = _candidate_for_synset(
synsets[0],
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
gain_need = 0.25 if aggressive else 0.45
if trial is not None and trial[1] >= gain_need:
chosen = synsets[0]
best_score = effective_min
elif aggressive and best_score >= effective_min * 0.5:
trial = _candidate_for_synset(
best_synset,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if trial is not None and trial[1] >= adj_adv_gain_floor:
chosen = best_synset
best_score = max(best_score, effective_min)
picked = (
_candidate_for_synset(
chosen,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if chosen is not None
else None
)
if not picked and polish and not aggressive:
# Same-synset lemmas are often empty (essential→?). Use WordNet
# related senses so polish can still densify wording safely.
related = _pick_related_modifier_candidate(
token,
doc,
synsets,
gain_floor=max(adj_adv_gain_floor, 0.25),
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if related:
best_score, replacement, simplicity_gain, chosen = related
picked = (replacement, simplicity_gain)
if not picked:
continue
replacement, simplicity_gain = picked
if aggressive and simplicity_gain < adj_adv_gain_floor and best_score < min_wsd:
continue
if polish and not aggressive and simplicity_gain < adj_adv_gain_floor:
continue
proposals.append(
(
best_score,
_impact_rank(token),
-simplicity_gain,
token.idx,
token.i,
replacement,
chosen,
)
)
if not proposals:
return LexicalResult(text=source, reason="no_confident_candidate")
if max_changes is None:
limit = dynamic_lexical_budget(source, polish=polish)
else:
limit = max(1, min(int(max_changes), 15))
if classical_strict and not classical_aggressive:
limit = min(limit, 2 if polish else 1)
elif classical_aggressive:
# Aggressive multi-pass: denser wording each hop.
limit = max(limit, 5 if polish else 3)
limit = min(limit, 10 if polish else 6)
if limit <= 0:
return LexicalResult(text=source, reason="budget_zero")
proposals.sort(key=lambda item: (-item[0], item[1], item[2], item[4]))
selected = proposals[:limit]
output = source
changes: list[LexicalChange] = []
for confidence, _impact, _gain, offset, token_index, replacement, synset in sorted(
selected,
key=lambda item: item[3],
reverse=True,
):
token = doc[token_index]
output = output[:offset] + replacement + output[offset + len(token.text) :]
if token_index > 0 and doc[token_index - 1].lower_ in {"a", "an"}:
prev = doc[token_index - 1]
needed = _article_for(replacement)
if prev.lower_ != needed:
article = needed.capitalize() if prev.text[:1].isupper() else needed
output = (
output[: prev.idx]
+ article
+ output[prev.idx + len(prev.text) :]
)
changes.append(
LexicalChange(
original=token.text,
replacement=replacement,
token_index=token_index,
lemma=token.lemma_,
synset_id=str(getattr(synset, "id", synset)),
confidence=round(confidence, 4),
)
)
if polish and not aggressive and (classical_aggressive or not classical_strict) and len(changes) < limit:
# Fill remaining polish budget with same-synset aggressive swaps only.
# Related-sense expansion stays single-pass to avoid chains like
# effective→strong→hard.
extra = refine_sentence(
output,
min_wsd=max(0.08, min_wsd * 0.8),
max_changes=limit - len(changes),
wordnet=resource,
aggressive=True,
polish=False,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if extra.changes and extra.text != output:
seen = {
(change.original.lower(), change.replacement.lower())
for change in changes
}
appended = False
for change in extra.changes:
pair = (change.original.lower(), change.replacement.lower())
if pair not in seen:
changes.append(change)
seen.add(pair)
appended = True
if appended:
output = extra.text
changes.sort(key=lambda change: change.token_index)
confidence = min(change.confidence for change in changes)
return LexicalResult(text=output, changes=changes, confidence=confidence)
def ensure_wording_change(
text: str,
*,
min_wsd: float = 0.08,
max_changes: int | None = None,
wordnet: Any | None = None,
polish: bool = False,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> LexicalResult:
"""Last-resort meaning-safe wording change for still-unchanged sentences."""
source = (text or "").strip()
if not source:
return LexicalResult(text=text, reason="empty")
if classical_strict and not classical_aggressive:
# Without MiniLM, do not force synonym swaps just to look different.
return LexicalResult(text=source, reason="classical_strict_no_ensure")
aggressive = refine_sentence(
source,
min_wsd=min_wsd,
max_changes=max_changes,
wordnet=wordnet,
aggressive=True,
polish=polish,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if aggressive.changes and aggressive.text.strip() != source:
aggressive.reason = aggressive.reason or "ensure_aggressive"
return aggressive
forced = _force_one_safe_swap(
source,
wordnet=wordnet,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if forced.changes and forced.text.strip() != source:
return forced
return LexicalResult(text=source, reason="ensure_unavailable")
def _force_one_safe_swap(
text: str,
*,
wordnet: Any | None = None,
classical_strict: bool = False,
classical_aggressive: bool = False,
) -> LexicalResult:
"""Pick one low-impact ADV/ADJ/VERB swap from the primary sense when possible."""
source = (text or "").strip()
nlp = get_nlp()
resource = wordnet if wordnet is not None else _get_wordnet()
if nlp is None or resource is None:
return LexicalResult(text=source, reason="resource_unavailable")
try:
doc = nlp(source)
except Exception:
return LexicalResult(text=source, reason="parse_failed")
stop_words = set(nlp.Defaults.stop_words)
ranked: list[tuple[int, float, float, int, str, Any]] = []
for token in doc:
if not _eligible(token, doc):
continue
if token.pos_ not in {"ADV", "ADJ", "VERB"}:
continue
try:
synsets = list(
resource.synsets(token.lemma_.lower(), pos=_POS_MAP[token.pos_])
)
except Exception:
continue
if not synsets:
continue
context = _context_terms(doc, token)
if token.pos_ == "VERB":
picked_verb = _pick_verb_candidate(
token,
doc,
synsets,
context,
stop_words,
effective_min=0.08,
gain_floor=0.30,
aggressive=True,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if not picked_verb:
continue
sense, replacement, simplicity_gain, synset = picked_verb
ranked.append(
(
_impact_rank(token),
-simplicity_gain,
-sense,
token.i,
replacement,
synset,
)
)
continue
# Prefer primary sense; for modifiers also try the best-scored sense.
candidates = [synsets[0]]
if len(synsets) > 1:
scored = sorted(
(
(_sense_score(s, context, stop_words), idx, s)
for idx, s in enumerate(synsets)
),
reverse=True,
)
if scored[0][2] not in candidates:
candidates.append(scored[0][2])
for synset in candidates:
picked = _candidate_for_synset(
synset,
token,
doc,
classical_strict=classical_strict,
classical_aggressive=classical_aggressive,
)
if not picked:
continue
replacement, simplicity_gain = picked
sense = _sense_score(synset, context, stop_words)
if simplicity_gain < 0.08 and sense < 0.05:
continue
ranked.append(
(
_impact_rank(token),
-simplicity_gain,
-sense,
token.i,
replacement,
synset,
)
)
break
if not ranked:
return LexicalResult(text=source, reason="no_force_candidate")
ranked.sort()
impact, _gain, _sense, token_index, replacement, synset = ranked[0]
token = doc[token_index]
output = source[: token.idx] + replacement + source[token.idx + len(token.text) :]
if token_index > 0 and doc[token_index - 1].lower_ in {"a", "an"}:
prev = doc[token_index - 1]
needed = _article_for(replacement)
if prev.lower_ != needed:
article = needed.capitalize() if prev.text[:1].isupper() else needed
output = (
output[: prev.idx] + article + output[prev.idx + len(prev.text) :]
)
change = LexicalChange(
original=token.text,
replacement=replacement,
token_index=token_index,
lemma=token.lemma_,
synset_id=str(getattr(synset, "id", synset)),
confidence=0.55,
)
return LexicalResult(
text=output,
changes=[change],
confidence=0.55,
reason="ensure_force_swap",
)