File size: 12,678 Bytes
e100026
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
# PuncAra — Arabic Punctuation Restoration Rules
# Extracted from PuncAra.py — preprocessing + postprocessing + chunking logic.
# All classes are imported by punctuation_service.py.

import re
import logging

logger = logging.getLogger(__name__)


def arabic_preprocessing(text: str) -> str:
    """Remove Arabic diacritics to normalize input for the model."""
    arabic_diacritics = re.compile(r'[\u064B-\u0652]')
    return re.sub(arabic_diacritics, '', text).strip()


def arabic_postprocessing(text: str) -> str:
    """
    Typographic cleanup and punctuation normalization after model inference.
    Handles: bracket spacing, duplicate marks, chunk-join artifacts, etc.
    """
    if not text:
        return text

    # 1. Protect numbers/fractions/time from incorrect conversion
    text = re.sub(r'(?<=\d),(?=\d)', '٪TEMP_COMMA٪', text)
    text = re.sub(r'(?<=\d):(?=\d)', '٪TEMP_COLON٪', text)

    # 2. Arabize typographic marks
    text = text.replace(',', '،').replace(';', '؛').replace('?', '؟')

    # 3. Fix internal spacing for brackets and Arabic quotes
    text = re.sub(r'\(\s+', '(', text)
    text = re.sub(r'\s+\)', ')', text)
    text = re.sub(r'\[\s+', '[', text)
    text = re.sub(r'\s+\]', ']', text)
    text = re.sub(r'«\s+', '«', text)
    text = re.sub(r'\s+»', '»', text)

    # 4. Remove repeated emotional marks (except ellipsis)
    text = re.sub(r'([،؛:!؟])\1+', r'\1', text)
    text = re.sub(r'\.{4,}', '...', text)

    # 5. Fix chunk-join contradictions
    text = re.sub(r'[،؛:]+([.!؟])', r'\1', text)
    text = re.sub(r'،؛|؛،', '؛', text)
    text = re.sub(r'([!؟])\.', r'\1', text)

    # 6. Remove stray leading punctuation
    text = re.sub(r'^[،؛:!؟. \t]+', '', text)

    # 7. Ensure single space after punctuation before text
    text = re.sub(r'([،؛:!؟.])(?=\S)', r'\1 ', text)

    # 8. Restore protected numbers
    text = text.replace('٪TEMP_COMMA٪', ',').replace('٪TEMP_COLON٪', ':')

    # 9. Attach punctuation to preceding word
    text = re.sub(r'\s+([،؛:!؟.])', r'\1', text)

    # 10. Collapse horizontal spaces only
    text = re.sub(r'[ \t]+', ' ', text).strip()
    return text


# ══════════════════════════════════════════════════════════════════════════════
# PUNCTUATION SAFETY LAYER — Pipeline Hardening v3.3
# ══════════════════════════════════════════════════════════════════════════════

ARABIC_PUNCT_CHARS = set('.,،؛؟!:;?!')
MAX_PUNCT_DELTA = 3
MAX_PUNCT_DELTA_SHORT = 1   # Stricter cap for short texts (≤2 words)
MAX_PUNCT_RATIO = 0.5       # max punctuation delta per word (multi-word diffs)


def _normalize_for_comparison(text: str) -> str:
    """
    Normalize Arabic for safe comparison.
    Prevents false rejection from hamza/alef/ya variants.
    """
    # Remove diacritics
    text = re.sub(r'[\u064B-\u0652]', '', text)
    # Fold hamza/alef variants: أ إ آ → ا
    text = re.sub(r'[أإآ]', 'ا', text)
    # Fold ya: ى → ي
    text = text.replace('ى', 'ي')
    # Fold ta marbuta: ة → ه (comparison only)
    text = text.replace('ة', 'ه')
    return text


def validate_punctuation_diff(diff: dict, full_text: str = '') -> bool:
    """
    Return True ONLY if the diff is a safe punctuation-only change.

    ALLOWED:
        - Inserting 1 punctuation mark (short text) or 1–3 (long text)
        - Replacing one punctuation mark with another
        - Adding terminal punctuation to any sentence (1+ words) that lacks it

    REJECTED:
        - Adding/deleting/duplicating Arabic words
        - Rewriting phrases
        - Excessive punctuation repetition (3+ consecutive identical)
        - Punctuation spam: delta/word_count > 0.5 (multi-word diffs)
        - Short text (≤2 words): delta > 1
        - Any diff: delta > MAX_PUNCT_DELTA
        - Adding terminal punctuation when text already ends with punct
    """
    original = diff.get('original', '')
    correction = diff.get('correction', '')

    # ── Rule 0 (FIX-01, updated FIX-30): Reject terminal punctuation injection ──
    # PuncAra-v1 unconditionally adds . or ؟ to every sentence.
    # This rule catches the pattern: "word" → "word." / "word؟" / "word،"
    # where the ONLY change is appending 1-2 terminal punctuation marks.
    #
    # FIX-30: Allow terminal punct for any text with at least 1 word that
    # doesn't already end with punctuation. Only block for:
    #   - Text that already has terminal punctuation
    #   - Text ending in an ellipsis (...)
    TERMINAL_PUNCT = set('.,،؛؟!:;?!')
    orig_stripped = original.rstrip()
    corr_stripped = correction.rstrip()
    if orig_stripped and corr_stripped:
        # Check if correction is just original + terminal punct
        orig_alpha_r0 = re.sub(r'[.,،؛؟!:;?\s]', '', original)
        corr_alpha_r0 = re.sub(r'[.,،؛؟!:;?\s]', '', correction)
        if (_normalize_for_comparison(orig_alpha_r0) ==
                _normalize_for_comparison(corr_alpha_r0)):
            # Same word content — check if only terminal punct was added
            orig_punct_end = sum(1 for c in original if c in TERMINAL_PUNCT)
            corr_punct_end = sum(1 for c in correction if c in TERMINAL_PUNCT)
            if corr_punct_end > orig_punct_end:
                # Only adding punctuation — check if it's at the END (terminal)
                orig_no_punct = re.sub(r'[.,،؛؟!:;?!]+$', '', original)
                corr_no_punct = re.sub(r'[.,،؛؟!:;?!]+$', '', correction)
                if _normalize_for_comparison(orig_no_punct.replace(' ', '')) == \
                   _normalize_for_comparison(corr_no_punct.replace(' ', '')):
                    # This is a pure terminal-punctuation addition.
                    # Decide whether to allow based on full text context.
                    # FIX-30: When full_text isn't provided (e.g. word-level diff
                    # calls), fall back to counting words in `original` instead of
                    # treating the count as 0 — that previously rejected every
                    # single-word diff regardless of the threshold below.
                    _word_count_source = full_text if full_text else original
                    _full_word_count = len(re.findall(
                        r'[\u0600-\u06FFa-zA-Z]+', _word_count_source
                    ))
                    _full_already_has_terminal = bool(
                        re.search(r'[.،؛؟!?!][\s]*$', full_text)
                    ) if full_text else False
                    # Also check for ellipsis (... at end)
                    _full_has_ellipsis = full_text.rstrip().endswith('...') if full_text else False

                    # FIX-30: Threshold lowered from 5 → 1. The docstring and the
                    # Phase 13 comment above both documented "3+ words" as the
                    # intended rule, while the code enforced 5 — and even single-
                    # word fragments ("اليوم" → "اليوم؟") are a legitimate terminal
                    # punctuation addition once we have at least one real word.
                    #
                    # FIX-31: Removed the FIX-29 exclamation/question-cue guard.
                    # It required an explicit interrogative word (هل/ماذا/متى/...)
                    # before allowing "؟" or "!" to be added, which rejected valid
                    # single-word terminal punctuation additions with no such cue
                    # (e.g. "اليوم" → "اليوم؟"). Terminal punctuation is now
                    # allowed regardless of cue words, as long as the remaining
                    # safety rules below (word count, duplicate terminal marks,
                    # ellipsis) still hold.
                    if _full_word_count >= 1 and not _full_already_has_terminal and not _full_has_ellipsis:
                        logger.info(
                            f"[PUNC-SAFETY] Allowed terminal punct for sentence "
                            f"({_full_word_count} words): "
                            f"'{original}' → '{correction}'"
                        )
                        # Fall through to remaining rules (don't return yet)
                    else:
                        # Already has terminal punct or ends in ellipsis → REJECT
                        logger.info(
                            f"[PUNC-SAFETY] TerminalPunctuationGuard triggered: removing trailing punctuation "
                            f"'{original}' → '{correction}'"
                        )
                        return False

    # ── Rule 0b (Batch 4): Reject punct insertion when original has no punctuation ──
    # If the original text has zero Arabic punctuation and the correction
    # only adds commas/semicolons (not at the very end), it's overcorrection.
    # This catches "already correct" texts that PuncAra sprinkles with commas.
    orig_punct_count_r0b = sum(1 for c in original if c in ARABIC_PUNCT_CHARS)
    if orig_punct_count_r0b == 0:
        corr_punct_count_r0b = sum(1 for c in correction if c in ARABIC_PUNCT_CHARS)
        if corr_punct_count_r0b > 0:
            # Only allow if adding a single period/question at the very end
            stripped_corr = correction.rstrip()
            if stripped_corr and stripped_corr[-1] in '.؟?!':
                # This is terminal punct (already handled by Rule 0)
                pass
            else:
                # Mid-sentence punct insertion on a clean sentence → reject
                logger.info(
                    f"[PUNC-SAFETY] Rejected mid-sentence punct insertion on clean text: "
                    f"'{original}' → '{correction}'"
                )
                return False

    # ── Rule 0c (Batch 4 + FIX-26): Reject punctuation rearrangement/substitution ──
    # When original already has punctuation and the correction merely MOVES,
    # SUBSTITUTES, or STACKS marks (e.g., ، → : or ، → ؛ or ؟ → ؟!), reject.
    # The PuncAra model should NOT replace or pile onto existing punctuation —
    # a sentence that already ends with punctuation must never get a second
    # mark added next to it.
    orig_punct_count_r0c = sum(1 for c in original if c in ARABIC_PUNCT_CHARS)
    corr_punct_count_r0c = sum(1 for c in correction if c in ARABIC_PUNCT_CHARS)
    if orig_punct_count_r0c > 0 and corr_punct_count_r0c > 0:
        # Both have punctuation — check if alpha content is the same
        orig_alpha_r0c = re.sub(r'[.,،؛؟!:;?\s]', '', original)
        corr_alpha_r0c = re.sub(r'[.,،؛؟!:;?\s]', '', correction)
        if _normalize_for_comparison(orig_alpha_r0c) == _normalize_for_comparison(corr_alpha_r0c):
            # Same word content, but punct changed — reject any punct modification,
            # whether it's a substitution or an addition on top of existing punct.
            logger.info(
                f"[PUNC-SAFETY] Rejected punct substitution/stacking: "
                f"'{original}' → '{correction}'"
            )
            return False

    # ── Rule 1: Alphabetic content must be identical after normalization ──
    orig_alpha = re.sub(r'[.,،؛؟!:;?\s]', '', original)
    corr_alpha = re.sub(r'[.,،؛؟!:;?\s]', '', correction)

    if _normalize_for_comparison(orig_alpha) != _normalize_for_comparison(corr_alpha):
        return False

    # ── Rule 2: Reject excessive repetition (3+ consecutive identical) ──
    if re.search(r'([.,،؛؟!:;?])\1{2,}', correction):
        return False

    # ── Shared computation for Rules 3–5 ──
    orig_punct_count = sum(1 for c in original if c in ARABIC_PUNCT_CHARS)
    corr_punct_count = sum(1 for c in correction if c in ARABIC_PUNCT_CHARS)
    punct_delta = max(0, corr_punct_count - orig_punct_count)
    word_count = len(re.findall(r'[\u0600-\u06FFa-zA-Z]+', correction)) or 1

    # ── Rule 3: Short-text hybrid cap (≤2 words → max 1 mark added) ──
    if word_count <= 2 and punct_delta > MAX_PUNCT_DELTA_SHORT:
        return False

    # ── Rule 4: Ratio-based spam protection (multi-word diffs) ──
    if word_count > 2 and punct_delta / word_count > MAX_PUNCT_RATIO:
        return False

    # ── Rule 5: Absolute delta cap ──
    if punct_delta > MAX_PUNCT_DELTA:
        return False

    return True