Spaces:
Configuration error
Configuration error
| import re | |
| from dataclasses import dataclass | |
| from typing import List | |
| import pandas as pd | |
| import config | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Data Model | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class QARecord: | |
| id: int | |
| question: str | |
| answer: str | |
| content: str | |
| tokens: List[str] | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Arabic Cleaning | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| ARABIC_DIACRITICS = re.compile(""" | |
| Ω | # Tashdid | |
| Ω | # Fatha | |
| Ω | # Tanwin Fath | |
| Ω | # Damma | |
| Ω | # Tanwin Damm | |
| Ω | # Kasra | |
| Ω | # Tanwin Kasr | |
| Ω | # Sukun | |
| Ω | |
| """, re.VERBOSE) | |
| def normalize_arabic(text): | |
| text = re.sub( | |
| "[Ψ₯Ψ£Ψ’Ψ§]", | |
| "Ψ§", | |
| text, | |
| ) | |
| text = re.sub( | |
| "Ω", | |
| "Ω", | |
| text, | |
| ) | |
| text = re.sub( | |
| "Ψ€", | |
| "Ω", | |
| text, | |
| ) | |
| text = re.sub( | |
| "Ψ¦", | |
| "Ω", | |
| text, | |
| ) | |
| text = re.sub( | |
| "Ψ©", | |
| "Ω", | |
| text, | |
| ) | |
| return text | |
| def strip_diacritics(text): | |
| return re.sub( | |
| ARABIC_DIACRITICS, | |
| "", | |
| text, | |
| ) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Cleaning | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def clean_text(text): | |
| text = str(text).strip() | |
| text = strip_diacritics(text) | |
| text = normalize_arabic(text) | |
| text = re.sub( | |
| r"\s+", | |
| " ", | |
| text, | |
| ) | |
| return text | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Tokenization | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def tokenize(text): | |
| text = clean_text(text) | |
| text = re.sub( | |
| r"[^\w\s]", | |
| " ", | |
| text, | |
| ) | |
| tokens = text.split() | |
| return tokens | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Load Excel | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def load_excel( | |
| path, | |
| question_col=None, | |
| answer_col=None, | |
| ): | |
| question_col = ( | |
| question_col | |
| or | |
| config.QUESTION_COL | |
| ) | |
| answer_col = ( | |
| answer_col | |
| or | |
| config.ANSWER_COL | |
| ) | |
| df = pd.read_excel(path) | |
| records = [] | |
| for idx, row in df.iterrows(): | |
| q = clean_text( | |
| row[question_col] | |
| ) | |
| a = clean_text( | |
| row[answer_col] | |
| ) | |
| # IMPORTANT: | |
| # index question + answer together | |
| content = f""" | |
| Ψ§ΩΨ³Ψ€Ψ§Ω: | |
| {q} | |
| Ψ§ΩΨ¬ΩΨ§Ψ¨: | |
| {a} | |
| """ | |
| tokens = tokenize(content) | |
| records.append( | |
| QARecord( | |
| id=idx, | |
| question=q, | |
| answer=a, | |
| content=content, | |
| tokens=tokens, | |
| ) | |
| ) | |
| return records |