| """Algorithmic rewrite quality gates (no word/phrase denylists)."""
|
|
|
| from __future__ import annotations
|
|
|
| import re
|
| from collections import Counter
|
|
|
| from app.pipeline.nlp import get_nlp
|
|
|
| _WORD = re.compile(r"[A-Za-z][A-Za-z']*")
|
| _MODAL = frozenset(
|
| {"can", "could", "will", "would", "may", "might", "should", "must", "shall"}
|
| )
|
|
|
|
|
| def _alpha_tokens(text: str) -> list[str]:
|
| return [m.group(0).lower() for m in _WORD.finditer(text or "")]
|
|
|
|
|
| def closed_compounds_not_split(original: str, candidate: str) -> bool:
|
| """Reject when a solid source token is opened into two words (teamwork→team work)."""
|
| source_tokens = _alpha_tokens(original)
|
| cand = re.sub(r"\s+", " ", (candidate or "").lower())
|
| cand_token_set = set(_alpha_tokens(candidate))
|
| for token in source_tokens:
|
| if len(token) < 6:
|
| continue
|
| if token in cand_token_set:
|
| continue
|
| for index in range(3, len(token) - 2):
|
| left, right = token[:index], token[index:]
|
| if f"{left} {right}" in cand:
|
| return False
|
| return True
|
|
|
|
|
| def modal_head_pos_preserved(original: str, candidate: str) -> bool:
|
| """Keep modal/aux clause heads stable in POS and lexical relatedness."""
|
| nlp = get_nlp()
|
| if nlp is None:
|
| return True
|
| try:
|
| source_doc = nlp(original or "")
|
| cand_doc = nlp(candidate or "")
|
| except Exception:
|
| return True
|
|
|
| def _modal_heads(doc) -> list[tuple[str, str, str]]:
|
| heads: list[tuple[str, str, str]] = []
|
| for token in doc:
|
| lemma = token.lemma_.lower()
|
| is_modal = lemma in _MODAL or (
|
| token.pos_ == "AUX" and token.dep_ in {"aux", "auxpass"}
|
| )
|
| if not is_modal:
|
| continue
|
| head = token.head
|
| if head.i == token.i:
|
| continue
|
| heads.append((lemma, head.pos_, head.lemma_.lower()))
|
| return heads
|
|
|
| def _verb_related(source_lemma: str, candidate_lemma: str) -> bool:
|
| if source_lemma == candidate_lemma:
|
| return True
|
| try:
|
| import wn
|
| from app.config import ENGINE_WORDNET_LEXICON
|
|
|
| wn.config.allow_multithreading = True
|
| resource = wn.Wordnet(ENGINE_WORDNET_LEXICON)
|
| synsets = list(resource.synsets(source_lemma, pos="v"))
|
| except Exception:
|
|
|
| return True
|
| for synset in synsets:
|
| try:
|
| words = {
|
| (word.lemma() or "").replace("_", " ").strip().lower()
|
| for word in synset.words()
|
| }
|
| except Exception:
|
| continue
|
| if candidate_lemma in words:
|
| return True
|
| return False
|
|
|
| source_heads = _modal_heads(source_doc)
|
| if not source_heads:
|
| return True
|
| cand_heads = _modal_heads(cand_doc)
|
| cand_by_modal: dict[str, list[tuple[str, str]]] = {}
|
| for modal, pos, lemma in cand_heads:
|
| cand_by_modal.setdefault(modal, []).append((pos, lemma))
|
| for modal, pos, lemma in source_heads:
|
| options = cand_by_modal.get(modal)
|
| if not options:
|
| continue
|
| matched = False
|
| for cand_pos, cand_lemma in options:
|
| if cand_pos != pos:
|
| continue
|
| if _verb_related(lemma, cand_lemma):
|
| matched = True
|
| break
|
| if not matched:
|
| return False
|
| return True
|
|
|
|
|
| def content_pos_balance_ok(original: str, candidate: str) -> bool:
|
| """Reject large shifts in content POS counts (noun/verb/adj)."""
|
| nlp = get_nlp()
|
| if nlp is None:
|
| return True
|
| try:
|
| source_doc = nlp(original or "")
|
| cand_doc = nlp(candidate or "")
|
| except Exception:
|
| return True
|
|
|
| def _counts(doc) -> Counter:
|
| return Counter(
|
| token.pos_
|
| for token in doc
|
| if token.is_alpha and not token.is_stop and token.pos_ in {"NOUN", "VERB", "ADJ"}
|
| )
|
|
|
| source = _counts(source_doc)
|
| cand = _counts(cand_doc)
|
| for pos in ("NOUN", "VERB", "ADJ"):
|
| if abs(source[pos] - cand[pos]) > 2:
|
| return False
|
| return True
|
|
|
|
|
| def substitution_pos_stable(
|
| sentence: str,
|
| token_index: int,
|
| replacement: str,
|
| ) -> bool:
|
| """Require the replaced span to keep the same coarse POS after re-parse."""
|
| nlp = get_nlp()
|
| if nlp is None:
|
| return True
|
| try:
|
| doc = nlp(sentence)
|
| except Exception:
|
| return True
|
| if token_index < 0 or token_index >= len(doc):
|
| return False
|
| token = doc[token_index]
|
| updated = sentence[: token.idx] + replacement + sentence[token.idx + len(token.text) :]
|
| try:
|
| new_doc = nlp(updated)
|
| except Exception:
|
| return True
|
|
|
| target = None
|
| for item in new_doc:
|
| if item.idx == token.idx:
|
| target = item
|
| break
|
| if target is None:
|
|
|
| for item in new_doc:
|
| if item.is_alpha and abs(item.idx - token.idx) <= max(1, len(replacement)):
|
| target = item
|
| break
|
| if target is None:
|
| return False
|
| return target.pos_ == token.pos_
|
|
|
|
|
| def meaning_ok(original: str, candidate: str, *, min_sim: float = 0.72) -> bool | None:
|
| """MiniLM meaning check when available; None means backend unavailable."""
|
| try:
|
| from app.pipeline.minilm import score_candidate
|
|
|
| scored = score_candidate(original, candidate)
|
| except Exception:
|
| return None
|
| if scored is None:
|
| return None
|
| return float(scored) >= min_sim
|
|
|
|
|
| def naturalness_reasons(original: str, candidate: str) -> list[str]:
|
| """Algorithmic naturalness failures (no phrase denylist)."""
|
| reasons: list[str] = []
|
| if not closed_compounds_not_split(original, candidate):
|
| reasons.append("compound_split")
|
| if not modal_head_pos_preserved(original, candidate):
|
| reasons.append("modal_pos_shift")
|
| if not content_pos_balance_ok(original, candidate):
|
| reasons.append("pos_balance")
|
| meaning = meaning_ok(original, candidate)
|
| if meaning is False:
|
| reasons.append("meaning_drop")
|
| return reasons
|
|
|