File size: 12,678 Bytes
e100026 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 | # PuncAra — Arabic Punctuation Restoration Rules
# Extracted from PuncAra.py — preprocessing + postprocessing + chunking logic.
# All classes are imported by punctuation_service.py.
import re
import logging
logger = logging.getLogger(__name__)
def arabic_preprocessing(text: str) -> str:
"""Remove Arabic diacritics to normalize input for the model."""
arabic_diacritics = re.compile(r'[\u064B-\u0652]')
return re.sub(arabic_diacritics, '', text).strip()
def arabic_postprocessing(text: str) -> str:
"""
Typographic cleanup and punctuation normalization after model inference.
Handles: bracket spacing, duplicate marks, chunk-join artifacts, etc.
"""
if not text:
return text
# 1. Protect numbers/fractions/time from incorrect conversion
text = re.sub(r'(?<=\d),(?=\d)', '٪TEMP_COMMA٪', text)
text = re.sub(r'(?<=\d):(?=\d)', '٪TEMP_COLON٪', text)
# 2. Arabize typographic marks
text = text.replace(',', '،').replace(';', '؛').replace('?', '؟')
# 3. Fix internal spacing for brackets and Arabic quotes
text = re.sub(r'\(\s+', '(', text)
text = re.sub(r'\s+\)', ')', text)
text = re.sub(r'\[\s+', '[', text)
text = re.sub(r'\s+\]', ']', text)
text = re.sub(r'«\s+', '«', text)
text = re.sub(r'\s+»', '»', text)
# 4. Remove repeated emotional marks (except ellipsis)
text = re.sub(r'([،؛:!؟])\1+', r'\1', text)
text = re.sub(r'\.{4,}', '...', text)
# 5. Fix chunk-join contradictions
text = re.sub(r'[،؛:]+([.!؟])', r'\1', text)
text = re.sub(r'،؛|؛،', '؛', text)
text = re.sub(r'([!؟])\.', r'\1', text)
# 6. Remove stray leading punctuation
text = re.sub(r'^[،؛:!؟. \t]+', '', text)
# 7. Ensure single space after punctuation before text
text = re.sub(r'([،؛:!؟.])(?=\S)', r'\1 ', text)
# 8. Restore protected numbers
text = text.replace('٪TEMP_COMMA٪', ',').replace('٪TEMP_COLON٪', ':')
# 9. Attach punctuation to preceding word
text = re.sub(r'\s+([،؛:!؟.])', r'\1', text)
# 10. Collapse horizontal spaces only
text = re.sub(r'[ \t]+', ' ', text).strip()
return text
# ══════════════════════════════════════════════════════════════════════════════
# PUNCTUATION SAFETY LAYER — Pipeline Hardening v3.3
# ══════════════════════════════════════════════════════════════════════════════
ARABIC_PUNCT_CHARS = set('.,،؛؟!:;?!')
MAX_PUNCT_DELTA = 3
MAX_PUNCT_DELTA_SHORT = 1 # Stricter cap for short texts (≤2 words)
MAX_PUNCT_RATIO = 0.5 # max punctuation delta per word (multi-word diffs)
def _normalize_for_comparison(text: str) -> str:
"""
Normalize Arabic for safe comparison.
Prevents false rejection from hamza/alef/ya variants.
"""
# Remove diacritics
text = re.sub(r'[\u064B-\u0652]', '', text)
# Fold hamza/alef variants: أ إ آ → ا
text = re.sub(r'[أإآ]', 'ا', text)
# Fold ya: ى → ي
text = text.replace('ى', 'ي')
# Fold ta marbuta: ة → ه (comparison only)
text = text.replace('ة', 'ه')
return text
def validate_punctuation_diff(diff: dict, full_text: str = '') -> bool:
"""
Return True ONLY if the diff is a safe punctuation-only change.
ALLOWED:
- Inserting 1 punctuation mark (short text) or 1–3 (long text)
- Replacing one punctuation mark with another
- Adding terminal punctuation to any sentence (1+ words) that lacks it
REJECTED:
- Adding/deleting/duplicating Arabic words
- Rewriting phrases
- Excessive punctuation repetition (3+ consecutive identical)
- Punctuation spam: delta/word_count > 0.5 (multi-word diffs)
- Short text (≤2 words): delta > 1
- Any diff: delta > MAX_PUNCT_DELTA
- Adding terminal punctuation when text already ends with punct
"""
original = diff.get('original', '')
correction = diff.get('correction', '')
# ── Rule 0 (FIX-01, updated FIX-30): Reject terminal punctuation injection ──
# PuncAra-v1 unconditionally adds . or ؟ to every sentence.
# This rule catches the pattern: "word" → "word." / "word؟" / "word،"
# where the ONLY change is appending 1-2 terminal punctuation marks.
#
# FIX-30: Allow terminal punct for any text with at least 1 word that
# doesn't already end with punctuation. Only block for:
# - Text that already has terminal punctuation
# - Text ending in an ellipsis (...)
TERMINAL_PUNCT = set('.,،؛؟!:;?!')
orig_stripped = original.rstrip()
corr_stripped = correction.rstrip()
if orig_stripped and corr_stripped:
# Check if correction is just original + terminal punct
orig_alpha_r0 = re.sub(r'[.,،؛؟!:;?\s]', '', original)
corr_alpha_r0 = re.sub(r'[.,،؛؟!:;?\s]', '', correction)
if (_normalize_for_comparison(orig_alpha_r0) ==
_normalize_for_comparison(corr_alpha_r0)):
# Same word content — check if only terminal punct was added
orig_punct_end = sum(1 for c in original if c in TERMINAL_PUNCT)
corr_punct_end = sum(1 for c in correction if c in TERMINAL_PUNCT)
if corr_punct_end > orig_punct_end:
# Only adding punctuation — check if it's at the END (terminal)
orig_no_punct = re.sub(r'[.,،؛؟!:;?!]+$', '', original)
corr_no_punct = re.sub(r'[.,،؛؟!:;?!]+$', '', correction)
if _normalize_for_comparison(orig_no_punct.replace(' ', '')) == \
_normalize_for_comparison(corr_no_punct.replace(' ', '')):
# This is a pure terminal-punctuation addition.
# Decide whether to allow based on full text context.
# FIX-30: When full_text isn't provided (e.g. word-level diff
# calls), fall back to counting words in `original` instead of
# treating the count as 0 — that previously rejected every
# single-word diff regardless of the threshold below.
_word_count_source = full_text if full_text else original
_full_word_count = len(re.findall(
r'[\u0600-\u06FFa-zA-Z]+', _word_count_source
))
_full_already_has_terminal = bool(
re.search(r'[.،؛؟!?!][\s]*$', full_text)
) if full_text else False
# Also check for ellipsis (... at end)
_full_has_ellipsis = full_text.rstrip().endswith('...') if full_text else False
# FIX-30: Threshold lowered from 5 → 1. The docstring and the
# Phase 13 comment above both documented "3+ words" as the
# intended rule, while the code enforced 5 — and even single-
# word fragments ("اليوم" → "اليوم؟") are a legitimate terminal
# punctuation addition once we have at least one real word.
#
# FIX-31: Removed the FIX-29 exclamation/question-cue guard.
# It required an explicit interrogative word (هل/ماذا/متى/...)
# before allowing "؟" or "!" to be added, which rejected valid
# single-word terminal punctuation additions with no such cue
# (e.g. "اليوم" → "اليوم؟"). Terminal punctuation is now
# allowed regardless of cue words, as long as the remaining
# safety rules below (word count, duplicate terminal marks,
# ellipsis) still hold.
if _full_word_count >= 1 and not _full_already_has_terminal and not _full_has_ellipsis:
logger.info(
f"[PUNC-SAFETY] Allowed terminal punct for sentence "
f"({_full_word_count} words): "
f"'{original}' → '{correction}'"
)
# Fall through to remaining rules (don't return yet)
else:
# Already has terminal punct or ends in ellipsis → REJECT
logger.info(
f"[PUNC-SAFETY] TerminalPunctuationGuard triggered: removing trailing punctuation "
f"'{original}' → '{correction}'"
)
return False
# ── Rule 0b (Batch 4): Reject punct insertion when original has no punctuation ──
# If the original text has zero Arabic punctuation and the correction
# only adds commas/semicolons (not at the very end), it's overcorrection.
# This catches "already correct" texts that PuncAra sprinkles with commas.
orig_punct_count_r0b = sum(1 for c in original if c in ARABIC_PUNCT_CHARS)
if orig_punct_count_r0b == 0:
corr_punct_count_r0b = sum(1 for c in correction if c in ARABIC_PUNCT_CHARS)
if corr_punct_count_r0b > 0:
# Only allow if adding a single period/question at the very end
stripped_corr = correction.rstrip()
if stripped_corr and stripped_corr[-1] in '.؟?!':
# This is terminal punct (already handled by Rule 0)
pass
else:
# Mid-sentence punct insertion on a clean sentence → reject
logger.info(
f"[PUNC-SAFETY] Rejected mid-sentence punct insertion on clean text: "
f"'{original}' → '{correction}'"
)
return False
# ── Rule 0c (Batch 4 + FIX-26): Reject punctuation rearrangement/substitution ──
# When original already has punctuation and the correction merely MOVES,
# SUBSTITUTES, or STACKS marks (e.g., ، → : or ، → ؛ or ؟ → ؟!), reject.
# The PuncAra model should NOT replace or pile onto existing punctuation —
# a sentence that already ends with punctuation must never get a second
# mark added next to it.
orig_punct_count_r0c = sum(1 for c in original if c in ARABIC_PUNCT_CHARS)
corr_punct_count_r0c = sum(1 for c in correction if c in ARABIC_PUNCT_CHARS)
if orig_punct_count_r0c > 0 and corr_punct_count_r0c > 0:
# Both have punctuation — check if alpha content is the same
orig_alpha_r0c = re.sub(r'[.,،؛؟!:;?\s]', '', original)
corr_alpha_r0c = re.sub(r'[.,،؛؟!:;?\s]', '', correction)
if _normalize_for_comparison(orig_alpha_r0c) == _normalize_for_comparison(corr_alpha_r0c):
# Same word content, but punct changed — reject any punct modification,
# whether it's a substitution or an addition on top of existing punct.
logger.info(
f"[PUNC-SAFETY] Rejected punct substitution/stacking: "
f"'{original}' → '{correction}'"
)
return False
# ── Rule 1: Alphabetic content must be identical after normalization ──
orig_alpha = re.sub(r'[.,،؛؟!:;?\s]', '', original)
corr_alpha = re.sub(r'[.,،؛؟!:;?\s]', '', correction)
if _normalize_for_comparison(orig_alpha) != _normalize_for_comparison(corr_alpha):
return False
# ── Rule 2: Reject excessive repetition (3+ consecutive identical) ──
if re.search(r'([.,،؛؟!:;?])\1{2,}', correction):
return False
# ── Shared computation for Rules 3–5 ──
orig_punct_count = sum(1 for c in original if c in ARABIC_PUNCT_CHARS)
corr_punct_count = sum(1 for c in correction if c in ARABIC_PUNCT_CHARS)
punct_delta = max(0, corr_punct_count - orig_punct_count)
word_count = len(re.findall(r'[\u0600-\u06FFa-zA-Z]+', correction)) or 1
# ── Rule 3: Short-text hybrid cap (≤2 words → max 1 mark added) ──
if word_count <= 2 and punct_delta > MAX_PUNCT_DELTA_SHORT:
return False
# ── Rule 4: Ratio-based spam protection (multi-word diffs) ──
if word_count > 2 and punct_delta / word_count > MAX_PUNCT_RATIO:
return False
# ── Rule 5: Absolute delta cap ──
if punct_delta > MAX_PUNCT_DELTA:
return False
return True
|