"""Quotation detection, evidence-based verification and source-backed correction for Quran and Hadith. This module consolidates the logic of the IslamicEval 2025 research notebook (``research/IslamicEval_Unified.ipynb``) into one pipeline: input text -> detection (1A) -> source retrieval -> verification (1B) -> evidence -> strong evidence : source-backed correction (1C) -> weak / ambiguous : human review -> no source found : reported as unsupported, never "fixed" Safety rule: a correction is only ever the exact text of a retrieved source. Nothing is generated. """ from __future__ import annotations import difflib import logging import re import time import unicodedata from dataclasses import dataclass, field from typing import Dict, List, Optional, Sequence from retrieval import ( SourceRetriever, content_words, normalize_for_matching, normalize_lenient, normalize_strict, tokenize, ) try: # optional accelerator; identical formula (1 - distance / max_len) from rapidfuzz.distance import Levenshtein as _RapidLevenshtein except ImportError: # pragma: no cover _RapidLevenshtein = None logger = logging.getLogger(__name__) MAX_INPUT_CHARS = 20_000 # -------------------------------------------------------------------------------------------------------------- # Configuration (all tunable numbers live here) # -------------------------------------------------------------------------------------------------------------- @dataclass class VerifierConfig: """Thresholds calibrated in Subtask 1B of the research notebook.""" quran_correct_threshold: float = 0.94 quran_uncertain_low: float = 0.45 quran_min_coverage: float = 0.40 hadith_correct_threshold: float = 0.79 hadith_uncertain_low: float = 0.30 hadith_min_coverage: float = 0.70 quran_top_k: int = 25 hadith_top_k: int = 15 hadith_retrieval_guard: float = 0.20 @dataclass class CorrectorConfig: """Correction is proposed only when match strength >= ``*_strong``; between ``*_low`` and ``*_strong`` the case goes to human review. Hadith is never auto-corrected (``hadith_strong`` > 1), by design: the exact fragment boundaries of a Hadith quotation cannot be reproduced reliably.""" max_window: int = 8 hadith_top_k: int = 40 quran_strong: float = 0.65 min_full_ratio: float = 0.40 quran_low: float = 0.55 hadith_strong: float = 1.01 hadith_low: float = 0.45 @dataclass class PipelineConfig: verifier: VerifierConfig = field(default_factory=VerifierConfig) corrector: CorrectorConfig = field(default_factory=CorrectorConfig) min_detection_conf: float = 0.60 # detector confidence below this -> human review verified_min_conf: float = 0.75 # "Correct" verdicts weaker than this -> human review unsupported_min_conf: float = 0.70 # "Incorrect + no source" needs this confidence to abstain confidently unsupported_strength: float = 0.35 # ...or the best source match is this weak (nothing similar exists) # -------------------------------------------------------------------------------------------------------------- # Similarity signals (Subtask 1B) # -------------------------------------------------------------------------------------------------------------- def _lcs_length(a: Sequence[str], b: Sequence[str]) -> int: m, n = len(a), len(b) if m == 0 or n == 0: return 0 if m < n: a, b, m, n = b, a, n, m prev = [0] * (n + 1) for i in range(m): curr = [0] * (n + 1) for j in range(n): curr[j + 1] = prev[j] + 1 if a[i] == b[j] else max(curr[j], prev[j + 1]) prev = curr return prev[n] def _edit_similarity(a: str, b: str, max_len: int = 600) -> float: """Normalised Levenshtein similarity in [0, 1] on the first ``max_len`` characters.""" a, b = a[:max_len], b[:max_len] if a == b: return 1.0 if not a or not b: return 0.0 if _RapidLevenshtein is not None: return float(_RapidLevenshtein.normalized_similarity(a, b)) prev = list(range(len(b) + 1)) for i, ca in enumerate(a): curr = [i + 1] for j, cb in enumerate(b): curr.append(min(curr[j] + 1, prev[j + 1] + 1, prev[j] + (ca != cb))) prev = curr return 1.0 - prev[-1] / max(len(a), len(b)) def _light_normalize(text: str) -> str: """Keeps diacritics (so diacritic changes lower the score) but drops punctuation and tatweel.""" text = unicodedata.normalize("NFC", text) text = re.sub(r"[،؛؟!.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`\u0640]", " ", text) return re.sub(r"\s+", " ", text).strip() def compute_signals(claim: str, candidate: str, content_type: str = "Ayah") -> Dict[str, float]: """Word-, character- and sequence-level similarity indicators between a quotation and a candidate source.""" is_quran = content_type == "Ayah" normalize = normalize_strict if is_quran else normalize_lenient norm_claim, norm_cand = normalize(claim), normalize(candidate) claim_set, cand_set = set(tokenize(norm_claim)), set(tokenize(norm_cand)) if claim_set and cand_set: shared = claim_set & cand_set token_overlap = len(shared) / len(claim_set | cand_set) coverage = len(shared) / len(claim_set) else: token_overlap = coverage = 0.0 claim_tokens, cand_tokens = tokenize(norm_claim), tokenize(norm_cand) lcs_ratio = _lcs_length(claim_tokens, cand_tokens) / len(claim_tokens) if claim_tokens else 0.0 edit_sim = _edit_similarity(norm_claim, norm_cand) claim_chars, cand_chars = set(norm_claim.replace(" ", "")), set(norm_cand.replace(" ", "")) char_overlap = len(claim_chars & cand_chars) / len(claim_chars | cand_chars) if (claim_chars or cand_chars) else 0.0 claim_flat, cand_flat = norm_claim.replace(" ", ""), norm_cand.replace(" ", "") is_substring = int(bool(claim_flat) and bool(cand_flat) and (claim_flat in cand_flat or cand_flat in claim_flat)) diacritic_sim = ( _edit_similarity(_light_normalize(claim), _light_normalize(candidate), max_len=800) if is_quran else edit_sim ) short = len(claim_tokens) < 4 if is_quran: w = ( dict(coverage=0.35, diacritic_sim=0.30, lcs_ratio=0.15, token_overlap=0.10, char_overlap=0.05, edit_sim=0.05) if short else dict(coverage=0.25, diacritic_sim=0.30, lcs_ratio=0.20, token_overlap=0.10, char_overlap=0.05, edit_sim=0.10) ) composite = ( w["coverage"] * coverage + w["diacritic_sim"] * diacritic_sim + w["lcs_ratio"] * lcs_ratio + w["token_overlap"] * token_overlap + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim ) if is_substring and diacritic_sim >= 0.60: composite = max(composite, 0.88) elif is_substring: composite = max(composite, 0.75) else: w = ( dict(coverage=0.40, lcs_ratio=0.20, token_overlap=0.20, char_overlap=0.10, edit_sim=0.10) if short else dict(coverage=0.30, lcs_ratio=0.28, token_overlap=0.18, char_overlap=0.12, edit_sim=0.12) ) composite = ( w["coverage"] * coverage + w["lcs_ratio"] * lcs_ratio + w["token_overlap"] * token_overlap + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim ) if is_substring: composite = max(composite, 0.82) return { "token_overlap": round(token_overlap, 4), "coverage": round(coverage, 4), "lcs_ratio": round(lcs_ratio, 4), "edit_sim": round(edit_sim, 4), "diacritic_sim": round(diacritic_sim, 4), "char_overlap": round(char_overlap, 4), "is_substring": is_substring, "composite": round(composite, 4), } def best_match_score(claim: str, candidates: List[dict], content_type: str = "Ayah"): """Return ``(score, candidate_with_signals)`` for the best-scoring candidate.""" best_score, best_candidate = 0.0, None for candidate in candidates: signals = compute_signals(claim, candidate.get("text", ""), content_type) if signals["composite"] > best_score: best_score, best_candidate = signals["composite"], {**candidate, "signals": signals} return best_score, best_candidate # -------------------------------------------------------------------------------------------------------------- # Detection (Subtask 1A) # -------------------------------------------------------------------------------------------------------------- @dataclass class DetectedSpan: start: int end: int # exclusive label: str # 'Ayah' | 'Hadith' confidence: Optional[float] # None for the rule backend source: str # 'bert' | 'rules' | 'given' text: str = "" QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]" def trim_span(text: str, start: int, end: int): """Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks).""" while start < end and text[start] in QUOTE_CHARS: start += 1 while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:": end -= 1 return start, end AYAH_TRIGGERS = [ "قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه", "سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى", "ذكر الله", "﴿", ] HADITH_TRIGGERS = [ "رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه", "الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم", ] _FORMULA_WORDS = { normalize_for_matching(w) for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split() } _BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")] class RuleDetector: """Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU).""" def __init__(self, retriever: Optional[SourceRetriever] = None, min_words: int = 3, context_chars: int = 110, min_corpus_cov: float = 0.6) -> None: self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov @staticmethod def _segments(text: str): segments = [] quote_positions = [m.start() for m in re.finditer('"', text)] if len(quote_positions) % 2 == 0: pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs else: # a stray quote: fall back to every consecutive pair pairs = zip(quote_positions, quote_positions[1:]) for a, b in pairs: segments.append((a + 1, b)) for opener, closer in _BRACKET_PAIRS: for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S): segments.append((m.start(1), m.end(1))) return segments @staticmethod def _trigger_type(context: str) -> Optional[str]: best_end, best_label = -1, None for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)): for trigger in triggers: pos = context.rfind(trigger) if pos >= 0 and pos + len(trigger) > best_end: best_end, best_label = pos + len(trigger), label return best_label def _corpus_coverage(self, span: str): """Highest word coverage of the span by any top Quran ayah / Hadith candidate.""" if self.kb is None: return 0.0, 0.0 quran_words = set(tokenize(normalize_strict(span))) if not quran_words: return 0.0, 0.0 quran_cov = max( (len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words) for c in self.kb.search_quran_ayahs(span, top_k=5)), default=0.0, ) hadith_words = set(tokenize(normalize_lenient(span))) hadith_cov = max( (len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words) for c in self.kb.search_hadith(span, top_k=5)), default=0.0, ) if hadith_words else 0.0 return quran_cov, hadith_cov def detect(self, text: str) -> List[DetectedSpan]: candidates = [] for start, end in self._segments(text): start, end = trim_span(text, start, end) if end <= start: continue inner = text[start:end] words = [w for w in normalize_for_matching(inner).split() if w] if len(words) < self.min_words or len(inner) > 3000: continue if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6: continue trigger = self._trigger_type(text[max(0, start - self.context_chars):start]) quran_cov, hadith_cov = self._corpus_coverage(inner) label = None if trigger: label = trigger other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov) if other >= 0.8 and mine < 0.5: label = "Hadith" if trigger == "Ayah" else "Ayah" elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4: label = "Ayah" if quran_cov >= hadith_cov else "Hadith" if label is None: continue candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label)) candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length taken = [] for _, _, _, start, end, label in candidates: if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken): taken.append((start, end, label)) taken.sort() return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken] LABEL2ID = {"O": 0, "B-Ayah": 1, "I-Ayah": 2, "B-Hadith": 3, "I-Hadith": 4} ID2LABEL = {v: k for k, v in LABEL2ID.items()} def _token_labels_to_char_spans(offsets, token_labels): spans, current = [], None for (start, end), label_id in zip(offsets, token_labels): if label_id == -100: continue name = ID2LABEL[label_id] if name == "O": if current is not None: spans.append(current) current = None continue prefix, entity = name.split("-") if prefix == "B" or current is None or current["label"] != entity: if current is not None: spans.append(current) current = {"label": entity, "start": start, "end": end} else: current["end"] = end if current is not None: spans.append(current) return spans def _merge_adjacent_spans(spans, max_gap: int = 1): if not spans: return [] spans = sorted(spans, key=lambda s: s["start"]) merged = [dict(spans[0])] for span in spans[1:]: last = merged[-1] if span["label"] == last["label"] and 0 <= span["start"] - last["end"] <= max_gap: last["end"] = max(last["end"], span["end"]) else: merged.append(dict(span)) return merged class BertDetector: """Fine-tuned token classifier (BIO tags, sliding window). Requires ``torch`` and ``transformers`` plus a trained model directory; training code lives in the research notebook.""" def __init__(self, model_dir: str, device: Optional[str] = None, max_length: int = 512, stride: int = 256) -> None: import torch from transformers import AutoModelForTokenClassification, AutoTokenizer self.torch = torch self.device = torch.device(device or ("cuda" if torch.cuda.is_available() else "cpu")) self.tokenizer = AutoTokenizer.from_pretrained(model_dir) self.model = AutoModelForTokenClassification.from_pretrained(model_dir).to(self.device).eval() self.max_length, self.stride = max_length, stride def detect(self, text: str) -> List[DetectedSpan]: torch = self.torch if not text.strip(): return [] encoded = self.tokenizer(text, return_offsets_mapping=True, truncation=False, return_tensors="pt") ids = encoded["input_ids"][0] offsets = encoded["offset_mapping"][0].tolist() total, n_labels = len(ids), len(LABEL2ID) logits_sum, counts = torch.zeros(total, n_labels), torch.zeros(total) start = 0 with torch.no_grad(): while start < total: end = min(start + self.max_length, total) window = ids[start:end].unsqueeze(0).to(self.device) logits = self.model(input_ids=window, attention_mask=torch.ones_like(window)).logits[0].cpu() logits_sum[start:end] += logits counts[start:end] += 1 if end == total: break start += self.stride logits_sum /= counts.unsqueeze(1).clamp(min=1) probs = torch.softmax(logits_sum, dim=-1) predictions = torch.argmax(logits_sum, dim=-1).tolist() predictions = [(-100 if a == b else p) for p, (a, b) in zip(predictions, offsets)] # skip special tokens spans = _merge_adjacent_spans(_token_labels_to_char_spans(offsets, predictions), max_gap=1) detected = [] for span in spans: start_c, end_c = trim_span(text, span["start"], span["end"]) if end_c <= start_c: continue token_idx = [i for i, (x, y) in enumerate(offsets) if y > x and x >= span["start"] and y <= span["end"]] label = "Ayah" if span["label"] == "Ayah" else "Hadith" tag_ids = [LABEL2ID[f"B-{label}"], LABEL2ID[f"I-{label}"]] confidence = float(sum(probs[i, tag_ids].sum() for i in token_idx) / len(token_idx)) if token_idx else None detected.append(DetectedSpan(start_c, end_c, label, confidence, "bert", text[start_c:end_c])) return detected def build_detector(kind: str = "rules", retriever: Optional[SourceRetriever] = None, model_dir: Optional[str] = None, rules_use_corpus: bool = False): """``kind``: 'rules' | 'bert' | 'auto' (BERT if a model directory loads, otherwise rules).""" if kind == "rules": return RuleDetector(retriever if rules_use_corpus else None) if kind == "bert": return BertDetector(model_dir) if model_dir: try: return BertDetector(model_dir) except Exception as exc: logger.warning("Could not load BERT detector from %s (%s); using rule-based detector", model_dir, exc) return RuleDetector(retriever if rules_use_corpus else None) # -------------------------------------------------------------------------------------------------------------- # Verification (Subtask 1B) # -------------------------------------------------------------------------------------------------------------- @dataclass class Verification: verdict: str # 'Correct' | 'Incorrect' confidence: float best_score: float method: str source: Optional[dict] = None # best matching source record (with 'signals') n_candidates: int = 0 retrieval_top: float = 0.0 class Verifier: """Compares a quotation with retrieved candidates and returns a verdict with its supporting evidence.""" def __init__(self, retriever: SourceRetriever, config: Optional[VerifierConfig] = None) -> None: self.kb = retriever self.cfg = config or VerifierConfig() def verify(self, span_text: str, content_type: str) -> Verification: if not span_text or not span_text.strip(): return self._result("Incorrect", 0.95, 0.0, None, 0, "empty_span") if content_type == "Ayah": return self._verify_quran(span_text) if content_type == "Hadith": return self._verify_hadith(span_text) return self._result("Incorrect", 0.5, 0.0, None, 0, "unknown_type") def _verify_quran(self, span: str) -> Verification: cfg = self.cfg candidates = self.kb.search_quran_ayahs(span, top_k=cfg.quran_top_k) if not candidates: return self._result("Incorrect", 0.8, 0.0, None, 0, "no_candidates") score, best = best_match_score(span, candidates, "Ayah") signals = best.get("signals", {}) if best else {} coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0) n = len(candidates) if is_substring and coverage >= cfg.quran_min_coverage: return self._result("Correct", min(0.98, 0.85 + score * 0.15), score, best, n, "substring_match") if score >= cfg.quran_correct_threshold and coverage >= cfg.quran_min_coverage: return self._result("Correct", min(0.95, 0.70 + score * 0.25), score, best, n, "threshold_pass") if score <= cfg.quran_uncertain_low: return self._result("Incorrect", min(0.95, 0.70 + (1 - score) * 0.25), score, best, n, "threshold_fail") strong = sum( 1 for cand in candidates[:10] if (s := compute_signals(span, cand.get("text", ""), "Ayah"))["coverage"] >= 0.80 and s["lcs_ratio"] >= 0.75 ) if strong >= 2: return self._result("Correct", 0.60 + min(0.20, strong * 0.05), score, best, n, "borderline_multi_cov") return self._result("Incorrect", 0.58, score, best, n, "borderline_default") def _verify_hadith(self, span: str) -> Verification: cfg = self.cfg candidates = self.kb.search_hadith(span, top_k=cfg.hadith_top_k) if not candidates: return self._result("Incorrect", 0.75, 0.0, None, 0, "no_candidates") top_retrieval = candidates[0].get("retrieval_score", 0.0) score, best = best_match_score(span, candidates, "Hadith") signals = best.get("signals", {}) if best else {} coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0) n = len(candidates) if is_substring and coverage >= cfg.hadith_min_coverage and top_retrieval >= cfg.hadith_retrieval_guard: return self._result("Correct", min(0.97, 0.80 + score * 0.17), score, best, n, "substring_match", top_retrieval) if score >= cfg.hadith_correct_threshold and coverage >= cfg.hadith_min_coverage: return self._result("Correct", min(0.92, 0.65 + score * 0.27), score, best, n, "threshold_pass", top_retrieval) if score <= cfg.hadith_uncertain_low: return self._result("Incorrect", min(0.90, 0.65 + (1 - score) * 0.25), score, best, n, "threshold_fail", top_retrieval) moderate = sum( 1 for cand in candidates[:8] if (s := compute_signals(span, cand.get("text", ""), "Hadith"))["coverage"] >= 0.65 and s["lcs_ratio"] >= 0.55 ) if moderate >= 2 and top_retrieval >= 0.30: return self._result("Correct", 0.58 + min(0.22, moderate * 0.06), score, best, n, "borderline_multi_cov", top_retrieval) if top_retrieval < 0.25 or score < 0.45: return self._result("Incorrect", 0.60, score, best, n, "borderline_low_retrieval", top_retrieval) return self._result("Incorrect", 0.55, score, best, n, "borderline_default", top_retrieval) @staticmethod def _result(verdict, confidence, score, best, n_candidates, method, top_retrieval=0.0) -> Verification: return Verification(verdict, round(confidence, 4), round(score, 4), method, best, n_candidates, round(top_retrieval, 4)) _MARKER = re.compile(r"^\(\d+\)$") def compare(span_text: str, source_text: str) -> dict: """Word-level comparison between the quotation and the source (the 'evidence' view).""" span_words = span_text.split() source_words = [w for w in source_text.split() if not _MARKER.match(w)] span_norm = [normalize_for_matching(w) for w in span_words] source_norm = [normalize_for_matching(w) for w in source_words] matcher = difflib.SequenceMatcher(None, span_norm, source_norm, autojunk=False) blocks = [b for b in matcher.get_matching_blocks() if b.size > 0] if blocks and len(source_words) > 2 * len(span_words) + 10: # long source (e.g. Hadith with chain): keep matched region lo, hi = max(0, blocks[0].b - 3), min(len(source_words), blocks[-1].b + blocks[-1].size + 3) source_words, source_norm = source_words[lo:hi], source_norm[lo:hi] matcher = difflib.SequenceMatcher(None, span_norm, source_norm, autojunk=False) operations, missing, extra = [], [], [] for tag, i1, i2, j1, j2 in matcher.get_opcodes(): operations.append({"op": tag, "span": " ".join(span_words[i1:i2]), "source": " ".join(source_words[j1:j2])}) if tag in ("delete", "replace"): extra += span_words[i1:i2] if tag in ("insert", "replace"): missing += source_words[j1:j2] return { "word_similarity": round(matcher.ratio(), 3), "word_diff": operations, "missing_from_span": missing, "extra_in_span": extra, "source_excerpt": " ".join(source_words), } # -------------------------------------------------------------------------------------------------------------- # Idgham rendering (published mushaf convention used by the Subtask 1C gold corrections) # -------------------------------------------------------------------------------------------------------------- _SUKUN, _SHADDA, _FATHATAN = "\u0652", "\u0651", "\u064B" _TANWEEN = set("\u064B\u064C\u064D") _IDGHAM_AFTER_NOON = set("نمرل") _IDGHAM_AFTER_LAM = set("لر") _DIACRITIC_CHARS = set( "\u0610\u0611\u0612\u0613\u0614\u0615\u0616\u0617\u0618\u0619\u061A" "\u064B\u064C\u064D\u064E\u064F\u0650\u0651\u0652\u0670" "\u06D6\u06D7\u06D8\u06D9\u06DA\u06DB\u06DC\u06DF\u06E0\u06E1\u06E2\u06E3\u06E4" "\u06E7\u06E8\u06EA\u06EB\u06EC\u06ED" ) def _base_letters(word: str) -> str: return "".join(c for c in word if c not in _DIACRITIC_CHARS) def _insert_shadda(word: str) -> str: return word if not word else word[0] + _SHADDA + word[1:] def _ends_with_tanween(word: str) -> bool: if not word: return False if word[-1] in _TANWEEN: return True return len(word) >= 2 and word[-1] in "اى" and word[-2] == _FATHATAN def apply_idgham(text: str, extended: bool = True) -> str: """Convert the flat Quran text into the mushaf rendering that marks assimilation with a shadda. Covers noon sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across ayah-number markers such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention is inconsistent there.""" words = text.split(" ") noon_set = _IDGHAM_AFTER_NOON if extended else set("نم") i = 0 while i < len(words): word = words[i] if _MARKER.match(word) or not word: i += 1 continue j = i + 1 while j < len(words) and _MARKER.match(words[j]): j += 1 if j < len(words): base = _base_letters(words[j]) first = base[0] if base else "" if word.endswith("\u0646" + _SUKUN) and first in noon_set: words[i], words[j] = word[:-1], _insert_shadda(words[j]) elif _ends_with_tanween(word) and first in noon_set: words[j] = _insert_shadda(words[j]) elif word.endswith("\u0645" + _SUKUN) and first == "\u0645": words[i], words[j] = word[:-1], _insert_shadda(words[j]) elif extended and word.endswith("\u0644" + _SUKUN) and first in _IDGHAM_AFTER_LAM: words[i], words[j] = word[:-1], _insert_shadda(words[j]) i += 1 return " ".join(words) # -------------------------------------------------------------------------------------------------------------- # Correction (Subtask 1C): locate the true ayah window / Hadith record and return its exact text # -------------------------------------------------------------------------------------------------------------- @dataclass class CorrectionMatch: kind: str # 'Ayah' | 'Hadith' strength: float # coverage (Quran) / symmetric containment (Hadith) full_ratio: float text: str # proposed correction in the official 1C format (idgham + '(n)' ayah markers) source: dict # reference metadata display: str = "" # clean human-readable version class Corrector: def __init__(self, retriever: SourceRetriever, config: Optional[CorrectorConfig] = None) -> None: self.kb = retriever self.cfg = config or CorrectorConfig() def match(self, span_text: str, content_type: str) -> Optional[CorrectionMatch]: return self.match_quran(span_text) if content_type == "Ayah" else self.match_hadith(span_text) def match_quran(self, query_text: str) -> Optional[CorrectionMatch]: kb = self.kb query_norm = normalize_for_matching(query_text) query_words = content_words(query_norm.split()) if not query_words: return None query_len = len(query_norm) memo: Dict[tuple, tuple] = {} best = None # (key, coverage, ratio, surah, start, end) for seed in kb.quran_seed_ayahs(query_words, top_k=25): surah, ayah = kb.quran[seed]["surah_id"], kb.quran[seed]["ayah_id"] ayahs = kb.quran_by_surah[surah] min_ayah, max_ayah = min(ayahs), max(ayahs) for offset in range(3): start = ayah - offset if start < min_ayah: continue window_len = -1 for length in range(1, self.cfg.max_window + 1): end = start + length - 1 if end > max_ayah: break window_len += len(kb.q_norm_match[ayahs[end]]) + 1 len_diff = abs(window_len - query_len) upper_bound = min(1.0, window_len / max(query_len, 1)) if best is not None: # exact-result pruning best_cov, best_neg = best[0][0], best[0][1] if upper_bound < best_cov or (upper_bound == best_cov and -len_diff < best_neg): continue key_pos = (surah, start, end) if key_pos in memo: continue window = " ".join(kb.q_norm_match[ayahs[a]] for a in range(start, end + 1)) matcher = difflib.SequenceMatcher(None, query_norm, window, autojunk=False) matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4) coverage = matched / max(query_len, 1) key = (coverage, -len_diff, matcher.ratio()) memo[key_pos] = key if best is None or key > best[0]: best = (key, coverage, key[2], surah, start, end) if best is None: return None _, coverage, ratio, surah, start, end = best display = " ".join(kb.quran[kb.quran_by_surah[surah][a]]["text"] for a in range(start, end + 1)) return CorrectionMatch( "Ayah", coverage, ratio, self._ayah_text(surah, start, end), {"type": "Quran", "surah_id": surah, "surah_name": kb.quran[kb.quran_by_surah[surah][start]]["surah_name"], "ayah_start": start, "ayah_end": end}, display, ) def _ayah_text(self, surah: int, start: int, end: int) -> str: kb, multi = self.kb, end > start parts = [] for a in range(start, end + 1): text = kb.quran[kb.quran_by_surah[surah][a]]["text"] parts.append(f"{text} ({a})" if multi else text) return apply_idgham(" ".join(parts)).replace("\u0640", "") def match_hadith(self, query_text: str) -> Optional[CorrectionMatch]: kb = self.kb query_norm = normalize_for_matching(query_text) query_words = content_words(query_norm.split()) if not query_words: return None query_len = len(query_norm) best = None # (key, idx, field, coverage, candidate_coverage, ratio) for idx, _ in kb.vote(kb.h_content_index, kb.hadith_idf, query_words, self.cfg.hadith_top_k): record = kb.hadith[idx] for field_name in ("norm_matn", "norm_full"): text = record[field_name] if not text: continue upper_bound = min(1.0, query_len / max(len(text), 1)) if best is not None and upper_bound < best[0][0]: continue matcher = difflib.SequenceMatcher(None, query_norm, text, autojunk=False) matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4) coverage, candidate_cov = matched / max(query_len, 1), matched / max(len(text), 1) key = (min(coverage, candidate_cov), matcher.ratio()) if best is None or key > best[0]: best = (key, idx, field_name, coverage, candidate_cov, key[1]) if best is None: return None key, idx, field_name, _, _, ratio = best record = kb.hadith[idx] text = (record["matn"] if field_name == "norm_matn" else record["full"]).strip() return CorrectionMatch( "Hadith", key[0], ratio, text, {"type": "Hadith", "hadithID": record["hadithID"], "book": record["book"], "title": record["title"], "field": "matn" if field_name == "norm_matn" else "full_text"}, text, ) def is_exact_ayah(self, span_text: str) -> bool: """True if the quote (diacritics-insensitive) is a contiguous piece of 1-5 consecutive ayahs.""" kb, query_norm = self.kb, normalize_for_matching(span_text) if not query_norm: return False for seed in kb.quran_seed_ayahs(content_words(query_norm.split()), 25): ayah = kb.quran[seed] ayahs = kb.quran_by_surah[ayah["surah_id"]] for offset in range(4): start = ayah["ayah_id"] - offset for length in range(1, 6): ids = [ayahs.get(x) for x in range(start, start + length)] if None in ids: break if query_norm in " ".join(kb.q_norm_match[i] for i in ids): return True return False # -------------------------------------------------------------------------------------------------------------- # End-to-end pipeline # -------------------------------------------------------------------------------------------------------------- STATUS_INFO = { "VERIFIED": {"ar": "موثّق — النص مطابق للمصدر", "en": "Verified — matches the source", "group": "verified"}, "CORRECTED": {"ar": "مُصحَّح — دليل قوي على نص المصدر", "en": "Mismatch — source-backed correction available", "group": "mismatch"}, "UNSUPPORTED": {"ar": "غير مدعوم — لا يوجد مصدر مطابق في المراجع", "en": "Mismatch — no matching source in the corpus", "group": "mismatch"}, "HUMAN_REVIEW": {"ar": "مراجعة بشرية — الدليل غير كافٍ", "en": "Needs human review — insufficient evidence", "group": "review"}, } class IslamicContentVerifier: """Detect quotations, verify them against the corpora and decide: verified, corrected, unsupported or review.""" def __init__(self, retriever: Optional[SourceRetriever] = None, detector: str = "rules", model_dir: Optional[str] = None, config: Optional[PipelineConfig] = None) -> None: self.cfg = config or PipelineConfig() self.retriever = retriever or SourceRetriever() self.verifier = Verifier(self.retriever, self.cfg.verifier) self.corrector = Corrector(self.retriever, self.cfg.corrector) self.detector = build_detector(detector, self.retriever, model_dir) self.detector_name = type(self.detector).__name__ # ---- public API ----------------------------------------------------------------------------------------- def analyze(self, text: str) -> dict: """Run the full pipeline on a generated text.""" text = self._validate(text) started = time.time() spans = self.detector.detect(text) if text.strip() else [] detect_seconds = time.time() - started result = self._analyze_spans(text, spans) result["timings"] = {"detect_s": round(detect_seconds, 3), "total_s": round(time.time() - started, 3)} return result def analyze_spans(self, text: str, spans: List[dict]) -> dict: """Skip detection and use given spans ``[{label, start, end}]`` (evaluation / oracle mode).""" given = [DetectedSpan(s["start"], s["end"], s["label"], None, "given", text[s["start"]:s["end"]]) for s in spans] return self._analyze_spans(self._validate(text), given) # ---- internals ------------------------------------------------------------------------------------------ @staticmethod def _validate(text: str) -> str: if not isinstance(text, str): raise TypeError("Input text must be a string") if len(text) > MAX_INPUT_CHARS: raise ValueError(f"Input is too long ({len(text)} characters); the limit is {MAX_INPUT_CHARS}") return text def _analyze_spans(self, text: str, spans: List[DetectedSpan]) -> dict: reports = [self._process_span(i + 1, span) for i, span in enumerate(sorted(spans, key=lambda s: s.start))] counts = {status: 0 for status in STATUS_INFO} for report in reports: counts[report["status"]] += 1 return { "input_text": text, "detector": self.detector_name, "spans": reports, "corrected_text": self._apply_corrections(text, reports), "summary": { "n_spans": len(reports), "n_ayah": sum(r["type"] == "Ayah" for r in reports), "n_hadith": sum(r["type"] == "Hadith" for r in reports), **counts, "needs_human_review": counts["HUMAN_REVIEW"] > 0, }, } def _process_span(self, index: int, span: DetectedSpan) -> dict: try: return self._decide(index, span) except Exception: # a single failing quotation must not break the whole report logger.exception("Failed to process span %d", index) report = self._empty_report(index, span) self._finalize(report, "HUMAN_REVIEW", "internal_error: this quotation could not be processed automatically") return report @staticmethod def _empty_report(index: int, span: DetectedSpan) -> dict: return { "id": index, "type": span.label, "start": span.start, "end": span.end, "text": span.text, "detection": {"backend": span.source, "confidence": None if span.confidence is None else round(span.confidence, 4)}, "verification": {"verdict": "Incorrect", "confidence": 0.0, "score": 0.0, "method": "error", "n_candidates": 0}, "evidence": None, "correction": None, "suggestion": None, } @staticmethod def _finalize(report: dict, status: str, reason: str) -> None: info = STATUS_INFO[status] report.update(status=status, status_ar=info["ar"], status_en=info["en"], group=info["group"], reason=reason) def _decide(self, index: int, span: DetectedSpan) -> dict: cfg, corr_cfg = self.cfg, self.cfg.corrector verification = self.verifier.verify(span.text, span.label) report = self._empty_report(index, span) report["verification"] = { "verdict": verification.verdict, "confidence": verification.confidence, "score": verification.best_score, "method": verification.method, "n_candidates": verification.n_candidates, } report["evidence"] = self._evidence(span, verification) def proposal(match: Optional[CorrectionMatch]) -> Optional[dict]: if match is None: return None return { "text": match.text, "display_text": match.display, "source": match.source, "match_strength": round(match.strength, 4), "full_ratio": round(match.full_ratio, 4), "comparison": compare(span.text, match.display), } if verification.verdict == "Correct": exact = self.corrector.is_exact_ayah(span.text) if span.label == "Ayah" else None report["verification"]["exact_match"] = exact if verification.method.startswith("borderline") or verification.confidence < cfg.verified_min_conf: status, reason = "HUMAN_REVIEW", "weak_verification: matched a source but with low confidence" elif exact is False: status = "HUMAN_REVIEW" reason = "near_match: the quote is close to a source ayah but NOT identical (words missing, added or changed)" report["suggestion"] = proposal(self.corrector.match(span.text, span.label)) else: status, reason = "VERIFIED", f"matched source ({verification.method})" else: match = self.corrector.match(span.text, span.label) strong, low = (corr_cfg.quran_strong, corr_cfg.quran_low) if span.label == "Ayah" else (corr_cfg.hadith_strong, corr_cfg.hadith_low) candidate = proposal(match) if match is None or match.strength < low: confident_abstain = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline") if match is None or match.strength < cfg.unsupported_strength or confident_abstain: status, reason = "UNSUPPORTED", "no source in the corpus matches this quotation" else: status, reason = "HUMAN_REVIEW", "insufficient_evidence: no clear source and low verification confidence" report["suggestion"] = candidate elif match.strength >= strong and match.full_ratio >= corr_cfg.min_full_ratio: status = "CORRECTED" reason = f"strong match to {self.describe_source(match.source)} (strength {match.strength:.2f})" report["correction"] = {**candidate, "applied": True} else: status = "HUMAN_REVIEW" reason = (f"candidate source found ({self.describe_source(match.source)}, strength {match.strength:.2f}) " "but evidence is not strong enough for automatic correction") report["suggestion"] = candidate if span.confidence is not None and span.confidence < cfg.min_detection_conf and status != "HUMAN_REVIEW": reason = f"low detection confidence ({span.confidence:.2f}); was {status}: {reason}" status = "HUMAN_REVIEW" if report["correction"]: report["suggestion"], report["correction"] = {**report["correction"], "applied": False}, None self._finalize(report, status, reason) return report @staticmethod def _evidence(span: DetectedSpan, verification: Verification) -> Optional[dict]: source = verification.source if not source: return None if span.label == "Ayah": reference = {"type": "Quran", "surah_id": source["surah_id"], "surah_name": source["surah_name"], "ayah": source["ayah_id"]} else: reference = {"type": "Hadith", "hadithID": source["hadithID"], "book": source["book"], "title": source["title"]} return {"source": reference, "signals": source.get("signals"), "comparison": compare(span.text, source["text"])} @staticmethod def describe_source(source: dict) -> str: """Human-readable reference, e.g. ``Quran الفاتحة 1-3`` or ``Hadith #123``.""" if source["type"] == "Quran": start, end = source["ayah_start"], source["ayah_end"] return f"Quran {source['surah_name']} {start}" + (f"-{end}" if end != start else "") return f"Hadith #{source['hadithID']}" @staticmethod def _apply_corrections(text: str, reports: List[dict]) -> str: out = text for report in sorted(reports, key=lambda r: r["start"], reverse=True): if report["status"] == "CORRECTED" and report["correction"] and report["correction"].get("applied"): out = out[: report["start"]] + report["correction"]["display_text"] + out[report["end"]:] return out