alfaz-browser / modules /study_audit.py
abedelbahnasy55's picture
Initial deploy: Flask + Gradio wrapper + assets
51d43b8 verified
Raw
History Blame Contribute Delete
24 kB
import re
import json
from pathlib import Path
from modules.term_workspace import TermWorkspace
from config import CITATION_DENSITY_TARGET, MIN_TOTAL_CITATIONS, MIN_STUDY_WORDS
# Section markers for splitting the study
SECTION_MARKERS = [
("tahdid", "تمهيد"),
("lugha", "المبحث الأول"),
("tarikh", "المبحث الثاني"),
("madloul", "المبحث الثالث"),
("muqarana", "المبحث الرابع"),
("khatima", "المبحث الخامس"),
("masadir", "قائمة المصادر"),
]
SECTION_NAMES = {
"tahdid": "التمهيد",
"lugha": "المبحث الأول: الدراسة اللغوية",
"tarikh": "المبحث الثاني: الدراسة الاصطلاحية",
"madloul": "المبحث الثالث: استعمال القطان",
"muqarana": "المبحث الرابع: الدراسة التطبيقية المقارنة",
"khatima": "المبحث الخامس: الخاتمة",
"masadir": "قائمة المصادر",
}
# Broad citation patterns — catches all common inline formats
CITATION_INLINE_PATTERN = (
r"\([^)]*(?:"
r"ت\s*\d+" # death year: (ت 852 هـ)
r"|ج\s*[:_]?\s*\d+" # volume: (ج1، ص10) or (ج:1)
r"|ص\s*[:_]?\s*\d+" # page: (ص15) or (ص:15)
r"|🔗" # link symbol
r"|https?://" # URL
r"|يحتاج توثيقاً" # needs documentation
r"|المصدر السابق" # previous source reference
r")[^)]*\)"
)
class StudyAuditor:
"""Audit a completed study for quality and compliance — section-aware."""
def __init__(self, term: str, category: str):
self.term = term
self.category = category
self.ws = TermWorkspace(term, category)
self.issues = []
self.stats = {}
def audit(self, text: str, draft: dict = None) -> dict:
self.issues = []
self.stats = {}
# Strip any existing audit report appendix first to ensure clean stats
text = re.split(r"\n## ملحق: تقرير الجودة", text)[0].strip()
self._count_citations(text)
self._check_citation_density(text)
self._check_hallucinated_links(text)
self._check_structure(text)
self._check_length(text)
self._check_language(text)
# New: section-aware analysis
sections = self._split_into_sections(text)
self._audit_sections(text, sections)
# New: raw material utilization
if draft:
self._check_raw_utilization(text, draft)
# New: cross-reference validation (book titles + narrators mentioned in study vs sources)
self._check_cross_references(text, draft)
passed = len(self.issues) == 0
severity = "PASS" if passed else "FAIL"
if self.issues:
severity = (
"WARN"
if all(i["severity"] == "warning" for i in self.issues)
else "FAIL"
)
result = {
"term": self.term,
"category": self.category,
"severity": severity,
"stats": self.stats,
"issues": self.issues,
"passed": passed,
}
self.ws.save_meta({"audit": result})
return result
def _split_into_sections(self, text: str) -> dict[str, str]:
"""Split study text into sections based on ## headings."""
sections = {}
lines = text.split("\n")
current_key = None
current_lines = []
for line in lines:
stripped = line.strip()
if stripped.startswith("## "):
# Save previous section
if current_key:
sections[current_key] = "\n".join(current_lines)
# Identify section
header_text = stripped[3:].strip()
current_key = None
for key, marker in SECTION_MARKERS:
if marker in header_text:
current_key = key
break
if current_key:
current_lines = [line]
else:
current_lines = [line]
elif current_key:
current_lines.append(line)
# Save last section
if current_key:
sections[current_key] = "\n".join(current_lines)
return sections
def _audit_sections(self, full_text: str, sections: dict[str, str]):
"""Per-section citation density and quality analysis."""
section_stats = {}
for key, marker in SECTION_MARKERS:
sec_text = sections.get(key, "")
if not sec_text:
section_stats[key] = {
"name": SECTION_NAMES.get(key, key),
"exists": False,
"words": 0,
"citations": 0,
"paragraphs": 0,
"paragraphs_without_citation": 0,
"density": 0.0,
}
continue
# Count citations in this section
citations = len(re.findall(CITATION_INLINE_PATTERN, sec_text))
# Count paragraphs
body = re.split(r"\n##\s*(?:قائمة\s+)?المصادر", sec_text)[0].strip()
paragraphs = [
p.strip()
for p in body.split("\n\n")
if p.strip()
and not p.strip().startswith("#")
and p.strip() not in ["---", "***", "___"]
and not p.strip().startswith("|")
]
paras_without = 0
for p in paragraphs:
if not re.search(CITATION_INLINE_PATTERN, p):
paras_without += 1
words = len(sec_text.split())
density = 1 - (paras_without / len(paragraphs)) if paragraphs else 0.0
section_stats[key] = {
"name": SECTION_NAMES.get(key, key),
"exists": True,
"words": words,
"citations": citations,
"paragraphs": len(paragraphs),
"paragraphs_without_citation": paras_without,
"density": round(density, 2),
}
# Flag sections with low citation density
if paragraphs and density < CITATION_DENSITY_TARGET and key != "masadir":
self.issues.append(
{
"type": "section_low_density",
"severity": "error",
"message": f"كثافة التوثيق منخفضة في {SECTION_NAMES.get(key, key)}: {density:.0%} ({paras_without}/{len(paragraphs)} فقرة بلا توثيق)",
"section": key,
}
)
# Flag sections that are too short
if key != "masadir" and 0 < words < 200:
self.issues.append(
{
"type": "section_too_short",
"severity": "warning",
"message": f"القسم {SECTION_NAMES.get(key, key)} قصير جداً: {words} كلمة",
"section": key,
}
)
self.stats["sections"] = section_stats
def _check_raw_utilization(self, text: str, draft: dict):
"""Check what percentage of raw material links appear in the study — per section."""
study_sections = self._split_into_sections(text)
section_util = {}
all_raw_links = set()
for sec_key, sec_data in draft.get("by_section", {}).items():
sec_links = set()
for item in sec_data.get("results", []):
link = item.get("link", "")
if link:
sec_links.add(link)
all_raw_links.add(link)
if not sec_links:
section_util[sec_key] = {"total": 0, "used": 0, "ratio": 0.0}
continue
sec_text = study_sections.get(sec_key, "")
sec_study_links = set(
re.findall(r"https://shamela\.ws/book/\d+/\d+", sec_text)
)
used = sec_links & sec_study_links
ratio = len(used) / len(sec_links) if sec_links else 0.0
section_util[sec_key] = {
"total": len(sec_links),
"used": len(used),
"ratio": round(ratio, 2),
}
self.stats["section_utilization"] = section_util
# Global stats for backward compatibility
all_study_links = set(re.findall(r"https://shamela\.ws/book/\d+/\d+", text))
global_used = all_raw_links & all_study_links
global_util = len(global_used) / len(all_raw_links) if all_raw_links else 0.0
self.stats["raw_links_total"] = len(all_raw_links)
self.stats["raw_links_used"] = len(global_used)
self.stats["raw_utilization"] = round(global_util, 2)
# Average per-section utilization for warning threshold
non_empty = [s for s in section_util.values() if s["total"] > 0]
avg_util = (
sum(s["ratio"] for s in non_empty) / len(non_empty) if non_empty else 0.0
)
if avg_util < 0.3 and non_empty:
self.issues.append(
{
"type": "low_raw_utilization",
"severity": "warning",
"message": f"نسبة استخدام المادة الخام منخفضة: {avg_util:.0%} (متوسط عبر الأقسام، {len(global_used)}/{len(all_raw_links)} رابط مستخدم)",
"details": f"تم استخدام {len(global_used)} رابط من أصل {len(all_raw_links)} في المادة الخام",
}
)
def _check_cross_references(self, text: str, draft: dict = None):
"""Validate that narrator names and book titles in the study match the raw material."""
if not draft:
return
# Collect known narrators from draft
known_narrators = set()
for sec_data in draft.get("by_section", {}).values():
for item in sec_data.get("results", []):
narrator = item.get("narrator", "").strip()
if narrator:
known_narrators.add(narrator)
# Collect known book titles from draft
known_books = set()
for sec_data in draft.get("by_section", {}).values():
for item in sec_data.get("results", []):
book = item.get("book", "").strip()
if book:
known_books.add(book)
# Check for narrator names mentioned in study but not in raw material
# Arabic name pattern: 2-4 words of Arabic chars
study_narrators = set()
for match in re.finditer(
r"(?:الراوي|الرواة|يقول|قال)\s+([^\s,،]+(?:\s+[^\s,،]+){0,2})", text
):
name = match.group(1).strip()
if len(name) > 3 and not any(
stop in name for stop in ["هذا", "هذه", "ذلك", "تلك", "الذي", "التي"]
):
study_narrators.add(name)
unvalidated_narrators = []
for name in study_narrators:
# Check if this name appears in any known narrator
found = any(name in kn or kn in name for kn in known_narrators)
if not found and len(name) > 5:
unvalidated_narrators.append(name)
if unvalidated_narrators:
self.stats["unvalidated_narrators"] = len(unvalidated_narrators)
self.issues.append(
{
"type": "unvalidated_narrators",
"severity": "warning",
"message": f"{len(unvalidated_narrators)} أسماء رواة في الدراسة لا تتطابق مع المادة الخام",
"details": unvalidated_narrators[:5],
}
)
self.stats["known_narrators"] = len(known_narrators)
self.stats["known_books"] = len(known_books)
def _count_citations(self, text: str):
numbered_pattern = r"\(\d+\)"
inline_citations = re.findall(CITATION_INLINE_PATTERN, text)
numbered_citations = re.findall(numbered_pattern, text)
total_inline = len(inline_citations)
total_numbered = len(numbered_citations)
self.stats["inline_citations"] = total_inline
self.stats["numbered_citations"] = total_numbered
self.stats["total_citations"] = total_inline + total_numbered
def _check_citation_density(self, text: str):
# Exclude the bibliography section (and anything after it) from paragraph density checks
body_text = re.split(r"\n##\s*(?:قائمة\s+)?المصادر", text)[0].strip()
paragraphs = [
p.strip()
for p in body_text.split("\n\n")
if p.strip()
and not p.strip().startswith("#")
and p.strip() not in ["---", "***", "___"]
and not p.strip().startswith("|")
]
total_paragraphs = len(paragraphs)
paragraphs_without_citation = 0
for p in paragraphs:
if not re.search(CITATION_INLINE_PATTERN, p):
paragraphs_without_citation += 1
self.stats["total_paragraphs"] = total_paragraphs
self.stats["paragraphs_without_citation"] = paragraphs_without_citation
if total_paragraphs > 0:
density = 1 - (paragraphs_without_citation / total_paragraphs)
self.stats["citation_density"] = round(density, 2)
if density < CITATION_DENSITY_TARGET:
self.issues.append(
{
"type": "low_citation_density",
"severity": "error",
"message": f"كثافة التوثيق منخفضة: {density:.0%} (المطلوب ≥ {CITATION_DENSITY_TARGET:.0%})",
"details": f"{paragraphs_without_citation}/{total_paragraphs} فقرات بلا توثيق",
}
)
if self.stats["total_citations"] < MIN_TOTAL_CITATIONS:
self.issues.append(
{
"type": "insufficient_citations",
"severity": "error",
"message": f"عدد التواقيع غير كافٍ: {self.stats['total_citations']} (الحد الأدنى: {MIN_TOTAL_CITATIONS})",
}
)
def _check_hallucinated_links(self, text: str):
links = re.findall(r"https://shamela\.ws/book/\d+/\d+", text)
valid_links = self.ws.raw_dir / "search_results.json"
if valid_links.exists():
draft = json.loads(valid_links.read_text(encoding="utf-8"))
known_links = set()
for sec_data in draft.get("by_section", {}).values():
for item in sec_data.get("results", []):
known_links.add(item.get("link", ""))
hallucinated = [l for l in links if l not in known_links]
self.stats["total_links"] = len(links)
self.stats["hallucinated_links"] = len(hallucinated)
if hallucinated:
self.issues.append(
{
"type": "hallucinated_links",
"severity": "error",
"message": f"تم اكتشاف {len(hallucinated)} روابط مختلقة",
"details": hallucinated[:5],
}
)
def _check_structure(self, text: str):
required = [
"تمهيد",
"المبحث الأول",
"المبحث الثاني",
"المبحث الثالث",
"المبحث الرابع",
"المبحث الخامس",
"قائمة المصادر",
]
missing = []
for section in required:
found = any(
section in line
for line in text.split("\n")
if line.strip().startswith("#")
)
if not found:
missing.append(section)
self.stats["required_sections"] = len(required)
self.stats["found_sections"] = len(required) - len(missing)
if missing:
self.issues.append(
{
"type": "missing_sections",
"severity": "error",
"message": f"أقسام مفقودة: {missing}",
}
)
def _check_length(self, text: str):
words = len(text.split())
chars = len(text)
lines = len(text.split("\n"))
self.stats["words"] = words
self.stats["chars"] = chars
self.stats["lines"] = lines
if words < MIN_STUDY_WORDS:
self.issues.append(
{
"type": "too_short",
"severity": "warning",
"message": f"الدراسة قصيرة جداً: {words} كلمة (المطلوب ≥ {MIN_STUDY_WORDS})",
}
)
def _check_language(self, text: str):
english_words = re.findall(r"\b[a-zA-Z]{3,}\b", text)
exclude_words = {
"sha",
"html",
"http",
"com",
"https",
"shamela",
"book",
"url",
"www",
"org",
"net",
}
english_words = [w for w in english_words if w.lower() not in exclude_words]
self.stats["english_words"] = len(english_words)
if english_words:
self.issues.append(
{
"type": "english_content",
"severity": "warning",
"message": f"تم اكتشاف {len(english_words)} كلمة إنجليزية",
"details": english_words[:10],
}
)
def print_report(self):
result = self.audit(
(self.ws.processed_dir / "study.md").read_text(encoding="utf-8")
if (self.ws.processed_dir / "study.md").exists()
else ""
)
print(f"\n{'=' * 60}")
print(f"تقرير Audit لللفظ: {self.term}")
print(f"{'=' * 60}")
print(f"الحالة: {'✅ PASS' if result['passed'] else '❌ FAIL'}")
print(f"{'=' * 60}")
print(f"\n📊 الإحصائيات العامة:")
for k, v in result["stats"].items():
if k != "sections":
print(f" {k}: {v}")
# Print section-level stats
sections = result["stats"].get("sections", {})
if sections:
print(f"\n📊 تفاصيل الأقسام:")
for key, sec in sections.items():
status = "✅" if sec["exists"] else "❌"
density_str = f"{sec['density']:.0%}" if sec["exists"] else "N/A"
print(
f" {status} {sec['name']}: {sec['words']} كلمة | {sec['citations']} توثيق | كثافة: {density_str}"
)
if result["issues"]:
print(f"\n⚠️ المشاكل ({len(result['issues'])}):")
for issue in result["issues"]:
icon = "❌" if issue["severity"] == "error" else "⚠️"
print(f" {icon} [{issue['type']}] {issue['message']}")
if "details" in issue:
for d in (
issue["details"][:3]
if isinstance(issue["details"], list)
else [issue["details"]]
):
print(f" → {d}")
else:
print(f"\n✅ لا توجد مشاكل!")
print(f"\n{'=' * 60}")
return result
def format_audit_report_markdown(audit_res: dict) -> str:
stats = audit_res.get("stats", {})
severity = audit_res.get("severity", "PASS")
severity_map = {
"PASS": "ناجح (مطابق للمنهجية)",
"WARN": "تنبيه (ملاحظات منهجية)",
"FAIL": "راسب (مخالف للمنهجية)",
}
severity_arabic = severity_map.get(severity, severity)
total_citations = stats.get("total_citations", 0)
density = stats.get("citation_density", 0.0)
density_percent = int(density * 100)
words = stats.get("words", 0)
total_paragraphs = stats.get("total_paragraphs", 0)
paragraphs_without_citation = stats.get("paragraphs_without_citation", 0)
found_sections = stats.get("found_sections", 0)
required_sections = stats.get("required_sections", 7)
issues = audit_res.get("issues", [])
if not issues:
issues_list = "* ✓ لا توجد أي مخالفات منهجية أو لغوية."
else:
issues_list = ""
for issue in issues:
icon = "❌" if issue.get("severity") == "error" else "⚠️"
issues_list += (
f"\n- {icon} **[{issue.get('type')}]** {issue.get('message')}"
)
if "details" in issue:
details = issue["details"]
if isinstance(details, list):
for d in details:
issues_list += f"\n - {d}"
else:
issues_list += f"\n - {details}"
# Section-level breakdown — no tables, use bullet lists
sections = stats.get("sections", {})
section_rows = ""
if sections:
section_rows = "\n### تفاصيل الأقسام:\n"
for key, sec in sections.items():
status = "✓" if sec.get("exists") else "✗"
density_val = f"{sec.get('density', 0):.0%}" if sec.get("exists") else "غير متاح"
section_rows += (
f"- **{sec.get('name', key)}**: الحالة {status}، "
f"الكلمات {sec.get('words', 0)}، "
f"التوثيقات {sec.get('citations', 0)}، "
f"الكثافة {density_val}\n"
)
# Raw utilization
raw_util = stats.get("raw_utilization")
raw_section = ""
if raw_util is not None:
raw_used = stats.get("raw_links_used", 0)
raw_total = stats.get("raw_links_total", 0)
raw_section = f"\n### استخدام المادة الخام:\n- نسبة الاستخدام: {raw_util:.0%} ({raw_used}/{raw_total} رابط)\n"
# Per-section breakdown — bullet list
sec_util = stats.get("section_utilization", {})
if sec_util:
for key, su in sec_util.items():
name = SECTION_NAMES.get(key, key)
ratio_str = f"{su['ratio']:.0%}"
raw_section += f"- **{name}**: المجموع {su['total']}، المستخدم {su['used']}، النسبة {ratio_str}\n"
report_md = f"""
## ملحق: تقرير الجودة المنهجية والأكاديمية
* **حالة التدقيق**: {severity_arabic}
* **إجمالي التوثيقات**: {total_citations}
* **كثافة التوثيق في الفقرات**: {density_percent}%
* **عدد الكلمات**: {words} كلمة
* **عدد الفقرات**: {total_paragraphs} فقرة
* **الفقرات الخالية من التوثيق**: {paragraphs_without_citation} فقرة
* **الأقسام المكتملة**: {found_sections} من {required_sections}
{section_rows}{raw_section}
### تفاصيل ملاحظات التدقيق المنهجي:
{issues_list}
"""
return report_md
if __name__ == "__main__":
import sys
if len(sys.argv) < 3:
print("Usage: python3 study_audit.py <term> <category>")
print(" category: tadeel or jarh")
sys.exit(1)
term = sys.argv[1]
category = sys.argv[2]
auditor = StudyAuditor(term, category)
auditor.print_report()