import sqlite3 import numpy as np import json import os import re import math from collections import defaultdict from concurrent.futures import ThreadPoolExecutor, as_completed VECTOR_DIM = 1024 INDEX_DB_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'search_index.db') AR_PHONETIC = { 'ا': 'أإآء', 'أ': 'اإآء', 'إ': 'اآأء', 'آ': 'اأإء', 'ء': 'اأإآ', 'ة': 'ه', 'ه': 'ة', 'ى': 'ي', 'ي': 'ىئ', 'ؤ': 'وء', 'و': 'ؤوء', 'ئ': 'يءى', 'س': 'ص', 'ص': 'س', 'ض': 'ظ', 'ظ': 'ض', 'ط': 'ت', 'ت': 'ط', 'ذ': 'زظض', 'ز': 'ذ', 'ظ': 'ضذ', 'ث': 'سص', 'س': 'ث', 'ح': 'ه', 'ه': 'ح', 'خ': 'غ', 'غ': 'خ', 'ق': 'ك', 'ك': 'ق', 'د': 'ذ', 'ذ': 'د', } AR_LETTERS = set('ابتثجحخدذرزسشصضطظعغفقكلمنهويئةءاأإآىؤ') AR2EN = { 'انسبشن': 'inception', 'سبع': 'seven', 'سي7ن': 'seven', 'سيفن': 'seven', 'براكنغ': 'breaking', 'باد': 'bad', 'سترينجر': 'stranger', 'ثينغز': 'things', 'سين': 'sen', 'انترستلر': 'interstellar', 'انترستيلر': 'interstellar', 'انتريستيلر': 'interstellar', 'فاست': 'fast', 'اند': 'and', 'فيوريوس': 'furious', 'ذي': 'the', 'برسوج': 'pursuit', 'هابينس': 'happiness', 'تقب': 'shawshank', 'تشانك': 'redemption', 'غود': 'good', 'فاذر': 'father', 'قذرنق': 'godfather', 'بالب': 'pulp', 'فكشن': 'fiction', 'فيكشن': 'fiction', 'دارك': 'dark', 'نايت': 'knight', 'رايزز': 'rises', 'فاتر': 'water', 'فور': 'for', 'إليفانتس': 'elephants', 'مسلسل': '', 'فيلم': '', 'مترجم': '', 'كامل': '', 'مدبلج': '', 'شبكة': '', 'نتفلكس': '', 'نتفليكس': '', 'الخامسة': '5', 'السادسة': '6', 'السابعة': '7', 'الثامنة': '8', 'التاسعة': '9', 'العاشرة': '10', 'الاول': '1', 'الثاني': '2', 'الثالث': '3', 'الرابع': '4', 'موسم': 'season', 'حلقة': 'episode', 'جزء': 'part', 'ثري ايديوتس': '3 Idiots', '3 بلاهة': '3 Idiots', 'عقل جميل': 'A Beautiful Mind', 'مكان هادئ': 'A Quiet Place', 'الهيبة': 'Al Hayba', 'اميركان هورور ستوري': 'American Horror Story', 'ملائكة وشياطين': 'Angels & Demons', 'انابيل': 'Annabelle', 'اكوامان': 'Aquaman', 'افاتار': 'Avatar', 'بداية باتمان': 'Batman Begins', 'بليد رانر': 'Blade Runner', 'بوجاك هورسمان': 'BoJack Horseman', 'حصان بوجاك': 'BoJack Horseman', 'بريكنج باد': 'Breaking Bad', 'اختراق': 'breaking', 'تشيرنوبل': 'Chernobyl', 'لعبة طفل': "Child's Play", 'صراع العمالقة': 'Clash of the Titans', 'قريب من البيت': 'Close to Home', 'كوبرا كاي': 'Cobra Kai', 'احتقار': 'Contempt', 'ديرديفيل': 'Daredevil', 'ديث نوت': 'Death Note', 'مفكرة الموت': 'Death Note', 'العاشق المحروم': 'Devdas', 'ديكستر': 'Dexter', 'داي هارد': 'Die Hard', 'المقاطعة 9': 'District 9', 'طبيبة الرومانسية': 'Doctor Romantic', 'جونية داركو': 'Donnie Darko', 'السقوط': 'Downfall', 'ايليت': 'Elite', 'النخبة': 'Elite', 'ايفل ديد': 'Evil Dead', 'فانتاستيك بيستس': 'Fantastic Beasts', 'السرعة والغضب': 'Fast & Furious', 'نادي القتال': 'Fight Club', 'فايت كلوب': 'Fight Club', 'فورست غامب': 'Forrest Gump', 'صائد الثعالب': 'Foxcatcher', 'فريندز': 'Friends', 'الأصدقاء': 'Friends', 'من': 'From', 'فروم': 'From', 'جيم اوف ثرونز': 'Game of Thrones', 'صراع العروش': 'Game of Thrones', 'غلاياديتور': 'Gladiator', 'المجالد': 'Gladiator', 'غودزيلا': 'Godzilla', 'غريز اناتومي': "Grey's Anatomy", 'هاكسو ريدج': 'Hacksaw Ridge', 'هانيبال': 'Hannibal', 'وحيد في المنزل': 'Home Alone', 'هاوس دي': 'House M.D', 'جو': 'john', 'انتريستيلر': 'Interstellar', 'بين النجوم': 'Interstellar', 'انترفيو وذ ذا فامباير': 'Interview with the Vampire', 'الفك المفترس': 'Jaws', 'جيغسو': 'Jigsaw', 'منشار الجرائم': 'Jigsaw', 'جون ويك': 'John Wick', 'جون ويك 4': 'John Wick: Chapter 4', 'رحلة ممتعة': 'Joy Ride', 'فيلم جوس': 'Juice', 'عصير': 'Juice', 'جومانجي': 'Jumanji', 'حديقة الديناصورات': 'Jurassic Park', 'كينغ كونغ': 'King Kong', 'مملكة السماء': 'Kingdom of Heaven', 'لا كازا دي بابل': 'La Casa de Papel', 'العدالة الناجزة': 'Law Abiding Citizen', 'حياة باي': 'Life of Pi', 'ليون ذا بروفيسونال': 'Léon: The Professional', 'ماد ماكس': 'Mad Max', 'تذكار': 'Memento', 'ميندهانتر': 'Mindhunter', 'المهمة المستحيلة': 'Mission: Impossible', 'فارس القمر': 'Moon Knight', 'مورتال كومبات': 'Mortal Kombat', 'ليل في المتحف': 'Night at the Museum', 'لا بلد للعجائز': 'No Country for Old Men', 'الآن أنت تراني': 'Now You See Me', 'اولد بوي': 'Oldboy', 'الفتى العجوز': 'Oldboy', 'اوبنهايمر': 'Oppenheimer', 'بيل ميت': 'Pale Blue Eye', 'بيكي بلايندرز': 'Peaky Blinders', 'قراصنة الكاريبي': 'Pirates of the Caribbean', 'برايمر': 'Primer', 'المنطلق الأول': 'Primer', 'بريسون بريك': 'Prison Break', 'خيال رخيص': 'Pulp Fiction', 'رجل المطر': 'Rain Man', 'ريال ستيل': 'Real Steel', 'فولاذ حقيقي': 'Real Steel', 'ريك اند مورتي': 'Rick and Morty', 'روبوكوب': 'RoboCop', 'رونالدو': 'Ronaldo', 'قائمة شيندلر': "Schindler's List", 'شيرلوك هولمز': 'Sherlock', 'الجزيرة الملعونة': 'Shutter Island', 'شوجون': 'Shōgun', 'الساموراي الأخير شوغون': 'Shōgun', 'سبايدرمان نو واي هوم': 'Spider-Man: No Way Home', 'سبيريتد اواي': 'Spirited Away', 'المخطوفة': 'Spirited Away', 'سبليت': 'Split', 'انفصام': 'Split', 'سترينجر ثينقز': 'Stranger Things', 'ساكسيشن': 'Succession', 'سوكايد سكواد': 'Suicide Squad', 'سوبرمان': 'Superman', 'ذا بير': 'The Bear', 'تأثير الفراشة': 'The Butterfly Effect', 'ذا كراون': 'The Crown', 'فارس الظلام': 'The Dark Knight', 'نهوض فارس الظلام': 'The Dark Knight Rises', 'المغادرون': 'The Departed', 'طارد الأرواح الشريرة': 'The Exorcist', 'الاب الروحي': 'The Godfather', 'فندق بودابست الكبير': 'The Grand Budapest Hotel', 'الميل الأخضر': 'The Green Mile', 'العاب الجوع': 'The Hunger Games', 'الرجل الأخضر': 'The Incredible Hulk', 'آخر من تبقى منا': 'The Last of Us', 'آخر الموهيكان': 'The Last of the Mohicans', 'سيد الخواتم': 'The Lord of the Rings', 'المرشح المنشوري': 'The Manchurian Candidate', 'الماتريكس': 'The Matrix', 'الآخرون': 'The Others', 'المنصة': 'The Platform', 'العظمة': 'The Prestige', 'صمت الحملان': 'The Silence of the Lambs', 'ذا سيمبسونز': 'The Simpsons', 'الحاسة السادسة': 'The Sixth Sense', 'الوصايا العشر': 'The Ten Commandments', 'المحطة': 'The Terminal', 'المبيد': 'The Terminator', 'المشتبه بهم المعتادون': 'The Usual Suspects', 'الموتى السائرون': 'The Walking Dead', 'ذا ووكينغ ديد': 'The Walking Dead', 'طريق العودة': 'The Way Back', 'ذا ويتشر': 'The Witcher', 'الذئب من وول ستريت': 'The Wolf of Wall Street', 'المستذئب': 'The Wolfman', 'ذيس ايس اس': 'This Is Us', 'تايتنك': 'Titanic', 'تايتانيك': 'Titanic', 'توب غان مافريك': 'Top Gun: Maverick', 'المتحولون': 'Transformers', 'ترو ديتكتيف': 'True Detective', 'فايكنجز': 'Vikings', 'والي': 'WALL-E', 'الذئاب لا تأكل اللحم': "Wolves Don't Eat Meat", } def normalize_text(text: str) -> str: text = text.strip().lower() text = re.sub(r'[_\-.+()\[\]{}!@#$%^&*,;:\'"<>?/\\|~`]', ' ', text) text = re.sub(r'\s+', ' ', text).strip() text = text.replace('إ', 'ا').replace('أ', 'ا').replace('آ', 'ا').replace('ئ', 'ي').replace('ؤ', 'و').replace('ة', 'ه').replace('ى', 'ي') return text AR2EN_NORM = {} for ak, av in AR2EN.items(): AR2EN_NORM[normalize_text(ak)] = av AR_DESC_EN = { 'رعب': 'horror', 'نفسي': 'psychological', 'خيال': 'fantasy fiction', 'خيال علمي': 'science fiction', 'علمي': 'scientific documentary', 'جريمة': 'crime', 'اغتيال': 'assassination', 'قتل': 'murder', 'قصة': 'story', 'وثائقي': 'documentary', 'فيلم': 'movie', 'مسلسل': 'series tv show', 'كوميديا': 'comedy', 'مضحك': 'funny', 'دراما': 'drama', 'اكشن': 'action', 'مغامرات': 'adventure', 'تشويق': 'thriller suspense', 'غموض': 'mystery', 'رومانسية': 'romance', 'حرب': 'war', 'تاريخي': 'historical', 'فضاء': 'space', 'كوكب': 'planet', 'مستقبل': 'future', 'زومبي': 'zombie', 'مصاصي': 'vampire', 'دماء': 'blood', 'شرطة': 'police', 'محقق': 'detective', 'جاسوسية': 'spy espionage', 'رياضة': 'sports', 'موسيقى': 'music', 'خارق': 'superhero supernatural', 'ابطال': 'superhero', 'رسوم': 'animated', 'كرتون': 'cartoon', 'عائلة': 'family', 'اطفال': 'children kids', 'مراهقين': 'teenagers teen', 'سجن': 'prison', 'عصابات': 'gang mafia', 'مافيا': 'mafia', 'كلاسيكي': 'classic', 'قديم': 'old vintage', 'مصر': 'egypt ancient egypt', 'فرعون': 'pharaoh egypt', 'حرب عالمية': 'world war', 'اصدقاء': 'friends', 'جيران': 'neighbors', 'مدرسة': 'school high school', 'جامعة': 'college university', 'طبيب': 'doctor medical', 'مستشفى': 'hospital medical', 'ساحر': 'magic wizard', 'مطاردة': 'chase pursuit', 'هروب': 'prison escape', 'محاكمة': 'court trial', 'قانون': 'law legal', 'موت': 'death', 'حياة': 'life', 'حب': 'love', 'خيانة': 'betrayal', 'انتقام': 'revenge', 'قراصنة': 'pirates', 'جزيرة': 'island', 'غابة': 'jungle forest', 'صحراء': 'desert', 'حيوانات': 'animals', 'كلب': 'dog', 'سيارة': 'car racing', 'سباق': 'racing', 'ملاكمة': 'boxing fighting', 'قتال': 'fighting martial arts', 'نينجا': 'ninja', 'ساموراي': 'samurai', 'ملك': 'king royal', 'امبراطورية': 'empire', 'سياسي': 'political', 'حكومة': 'government', 'صحفي': 'journalist news', 'ديناصور': 'dinosaur', 'وحش': 'monster creature', 'شبح': 'ghost supernatural', 'ارواح': 'ghost spirits', 'مؤامرة': 'conspiracy', 'نهاية': 'apocalypse end of world', 'كوارث': 'disaster', 'تنين': 'dragon', 'سيف': 'sword medieval', 'عصور وسطى': 'medieval middle ages', 'مصارعة': 'wrestling', 'بطولة': 'championship tournament', 'مشاهير': 'celebrity famous', 'عن': 'about', 'عالم': 'world space', 'يدخل': 'enters into', 'داخل': 'inside into', 'ثقب': 'hole', 'أسود': 'black', 'ينقذ': 'saves rescues', 'بنته': 'his daughter', 'كيميائي': 'chemist chemistry', 'يصنع': 'makes produces', 'مواد': 'substances materials', 'ممنوعة': 'illegal forbidden banned', 'مع': 'with', 'طالب': 'student', 'قديم': 'old former', 'سفينة': 'ship boat', 'ضخمة': 'huge giant massive', 'تصطدم': 'crashes hits collides', 'بجبل': 'mountain iceberg', 'جليدي': 'ice frozen', 'وتغرق': 'sinks drowning', 'شخص': 'person man', 'يعيش': 'lives', 'محمية': 'nature reserve dome', 'ويعتقد': 'believes thinks', 'أنه': 'he is', 'معزولة': 'isolated remote', 'ويمثلون': 'acting pretend', 'عليه': 'on him', 'سرقة': 'robbery heist', 'بنك': 'bank', 'إسبانيا': 'spain spanish', 'بذكاء': 'cleverly smartly', 'بروفيسور': 'professor', 'وأقنعة': 'masks', 'دالي': 'dali', 'غني': 'rich wealthy', 'يرتدي': 'wears dresses', 'ملابس': 'clothes costume', 'خفاش': 'bat', 'ويحارب': 'fights battles', 'الجوكر': 'joker', 'صراع': 'conflict war', 'عروش': 'thrones', 'ممالك': 'kingdoms', 'وتنانين': 'dragons', 'وموتى': 'dead zombies', 'سائرون': 'walking', 'أحلام': 'dreams', 'وسرقة': 'stealing', 'أفكار': 'ideas thoughts', 'العقل': 'the mind brain', 'أسد': 'lion', 'صغير': 'small young little', 'يموت': 'dies', 'والده': 'his father', 'ويصبح': 'becomes', 'الغابة': 'forest jungle', 'ملاكم': 'boxer fighter', 'ضعيف': 'weak poor', 'يتحدى': 'challenges defies', 'بطل': 'champion hero', 'أسطورة': 'legend', 'على': 'on', 'دراجات': 'bicycles bikes', 'يواجهون': 'face encounter', 'وحوش': 'monsters creatures', 'مقلوب': 'upside down inverted', 'تائهة': 'lost stranded', 'والذكاء': 'intelligence ai', 'الاصطناعي': 'artificial', 'ينقلب': 'turns rebels', 'الطاقم': 'crew', 'قاتل': 'killer assassin hitman', 'مأجور': 'hired paid', 'يقتل': 'kills murders', 'الكل': 'everyone everybody', 'بسبب': 'because of', 'كلبه': 'his dog puppy', 'عبقري': 'genius brilliant', 'يهرب': 'escapes flees', 'أخوه': 'his brother', 'محصن': 'fortified maximum security', 'تاتو': 'tattoo', 'الناس': 'people', 'عايشة': 'living', 'محاكاة': 'simulation virtual', 'كمبيوتر': 'computer', 'والكبسولة': 'the pill capsule', 'الحمراء': 'red', 'يعاني': 'suffers', 'فقدان': 'loss losing', 'ذاكرة': 'memory', 'مؤقت': 'temporary short term', 'ويبحث': 'searches looks for', 'زوجته': 'his wife', 'بالصور': 'with photos pictures', 'كوري': 'korean', 'ألعاب': 'games', 'شعبية': 'popular children', 'والمخسر': 'the loser', 'أجل': 'for sake', 'المال': 'money', 'تنتقل': 'moves relocates', 'لمنزل': 'to a house', 'مسكون': 'haunted', 'وتستعين': 'seeks help calls', 'بخبراء': 'experts', 'اد': 'ed', 'ولورين': 'lorraine', 'رائد': 'astronaut', 'وحيدا': 'alone', 'المريخ': 'mars', 'ويزرع': 'grows plants farms', 'البطاطس': 'potatoes', 'موظف': 'employee clerk worker', 'يكتشف': 'discovers realizes', 'شخصية': 'character', 'ثانوية': 'secondary side', 'لعبة': 'game', 'فيديو': 'video', 'شاب': 'young man boy', 'فقير': 'poor', 'نيويورك': 'new york', 'ارتدى': 'wore puts on', 'بدلة': 'suit costume', 'حديد': 'iron metal', 'وطار': 'flew flies', 'سماء': 'sky', 'يستيقظ': 'wakes up', 'نفسه': 'the same', 'ويكتشف': 'discovers finds', 'قدرات': 'abilities powers', 'يستخدمون': 'they use', 'قدراتهم': 'their powers', 'لمحاربة': 'to fight against', 'الشر': 'evil', 'يتيم': 'orphan', 'خزانة': 'closet wardrobe', 'مدخل': 'entrance portal', 'سحري': 'magical wizard', 'خالته': 'his aunt', 'وعمه': 'and uncle', 'المستذئبين': 'werewolves', 'مصاصي': 'vampire', 'الدماء': 'blood', 'ساحر': 'wizard magician', 'الحرب': 'war', 'العالمية': 'world', 'ألمانية': 'german nazi', 'ثنائي': 'duo pair', 'شرطي': 'police officer cops', 'رفقاء': 'partners buddies', 'فاخرة': 'luxury expensive', 'يسافرون': 'travel journey', 'لحظات': 'moments', 'ثلاثة': 'three', 'رجال': 'men', 'يخطفون': 'they kidnap', 'عروس': 'bride', 'ليلة': 'night', 'زفافها': 'her wedding', 'ويطلبون': 'demand asking', 'فدية': 'ransom', 'لكن': 'but', 'الأمور': 'things', 'تتعقد': 'get complicated', 'سيدة': 'lady woman', 'مسنة': 'elderly old', 'تفقد': 'loses misplaces', 'قطتها': 'her cat', 'فتحول': 'turns transforms', 'المنزل': 'the house neighborhood', 'للعثور': 'to find search', 'عليها': 'for her', 'يكشف': 'reveals uncovers', 'كبيرة': 'big large huge', 'أمريكي': 'american', 'يحلم': 'dreams', 'يكون': 'to be', 'بطلا': 'hero champion', 'لكن': 'but', 'الواقع': 'reality', 'مختلف': 'different', 'مجموعة': 'group team', 'كل': 'each every', 'منهم': 'of them', 'لديه': 'has', 'حلم': 'dream', 'ويعملون': 'and work', 'معا': 'together', 'لتحقيقه': 'to achieve it', } def tokenize(text: str) -> list: return normalize_text(text).split() def compute_tfidf(docs: list) -> dict: N = len(docs) df = defaultdict(int) for doc in docs: seen = set() for t in doc: if t not in seen: df[t] += 1 seen.add(t) return {t: math.log((N + 1) / (f + 0.5) + 1.0) for t, f in df.items()} class BM25Index: def __init__(self, k1: float = 1.5, b: float = 0.75): self.k1 = k1 self.b = b self.doc_count = 0 self.avg_dl = 0.0 self.idf = {} self.term_doc_freq = defaultdict(lambda: defaultdict(int)) self.doc_lens = [] self.doc_fields = [] self.id_map = [] def build(self, docs: list, idf: dict): self.doc_count = len(docs) self.idf = idf self.doc_lens = [len(d) for d in docs] self.avg_dl = sum(self.doc_lens) / max(self.doc_count, 1) for di, doc in enumerate(docs): tf = defaultdict(int) for t in doc: tf[t] += 1 for t, f in tf.items(): self.term_doc_freq[t][di] = f def search(self, query: list, top_k: int = 50) -> list: scores = defaultdict(float) for qt in query: if qt not in self.idf: continue idf = self.idf[qt] for di, freq in self.term_doc_freq.get(qt, {}).items(): dl = self.doc_lens[di] num = freq * (self.k1 + 1) den = freq + self.k1 * (1 - self.b + self.b * dl / max(self.avg_dl, 1)) scores[di] += idf * num / max(den, 1e-10) ranked = sorted(scores.items(), key=lambda x: -x[1]) return ranked[:top_k] class SpellingCorrector: def __init__(self): self.vocab = set() self.word_freq = defaultdict(int) self.title_vocab = set() self.normalized_vocab = {} self.correction_cache = {} def add_titles(self, titles: list): for t in titles: if not t: continue nt = normalize_text(t) self.normalized_vocab[nt] = t self.title_vocab.add(t) self.title_vocab.add(nt) for w in t.split(): wc = w.strip().lower() if wc: self.vocab.add(wc) self.word_freq[wc] += 1 if re.search(r'[\u0600-\u06FF]', wc): for ar_word, en_word in AR2EN.items(): if ar_word in wc: self.vocab.add(en_word) for w in nt.split(): if w: self.vocab.add(w) self.word_freq[w] += 1 self._vocab_by_len = {} for vw in self.vocab: self._vocab_by_len.setdefault(len(vw), []).append(vw) for length in self._vocab_by_len: self._vocab_by_len[length].sort(key=lambda x: -self.word_freq.get(x, 0)) self._vocab_by_len[length] = self._vocab_by_len[length][:500] def add_words(self, words: list): for w in words: wc = w.strip().lower() if wc: self.vocab.add(wc) self.word_freq[wc] += 1 def _jaccard_ngrams(self, a: str, b: str, n: int = 3) -> float: if len(a) < n or len(b) < n: return 0.0 ag = set(a[i:i+n] for i in range(len(a)-n+1)) bg = set(b[i:i+n] for i in range(len(b)-n+1)) if not ag or not bg: return 0.0 return len(ag & bg) / len(ag | bg) def _lev_ratio(self, a: str, b: str) -> float: if not a or not b: return 0.0 la, lb = len(a), len(b) if abs(la - lb) > max(3, la // 2): return 0.0 d = [[0]*(lb+1) for _ in range(la+1)] for i in range(la+1): d[i][0] = i for j in range(lb+1): d[0][j] = j for i in range(1, la+1): for j in range(1, lb+1): cost = 0 if a[i-1] == b[j-1] else 1 d[i][j] = min(d[i-1][j]+1, d[i][j-1]+1, d[i-1][j-1]+cost) return 1.0 - (d[la][lb] / max(la, lb)) def _ar_phonetic_key(self, word: str) -> str: res = [] for ch in word: key = ch for base, group in AR_PHONETIC.items(): if ch in group or ch == base: key = base break res.append(key) return ''.join(res) def correct_word(self, word: str) -> tuple: w = word.strip().lower() if not w: return (word, 1.0, False) if w in self.vocab: return (word, 1.0, False) if w in self.correction_cache: cached = self.correction_cache[w] return (cached[0], cached[1], True) if w in AR2EN: mapped = AR2EN[w] result = (mapped, 0.95, True) if mapped else (word, 0.0, False) self.correction_cache[w] = (result[0], result[1]) return result if re.search(r'[\u0600-\u06FF]', w): for ar_word, en_word in AR2EN.items(): if ar_word in w: rem = w.replace(ar_word, '').strip() if rem: rem_c, _, _ = self.correct_word(rem) result = ((en_word + ' ' + rem_c).strip(), 0.8, True) self.correction_cache[w] = (result[0], result[1]) return result result = (en_word, 0.85, True) self.correction_cache[w] = (result[0], result[1]) return result if len(w) < 4: self.correction_cache[w] = (word, 0.0, False) return (word, 0.0, False) candidates = [] wl = len(w) for delta in range(0, 2): for l in (wl - delta, wl + delta): if l <= 0 or l > 30: continue for vw in self._vocab_by_len.get(l, []): if abs(len(vw) - wl) > 2: continue lr = self._lev_ratio(w, vw) if lr >= 0.5: jacc = self._jaccard_ngrams(w, vw) score = lr * 0.6 + jacc * 0.4 + min(self.word_freq.get(vw, 0) / 10, 0.1) if score > 0.6: candidates.append((vw, score)) if candidates: break if not candidates: pk = self._ar_phonetic_key(w) for vw in self.vocab: if self._ar_phonetic_key(vw) == pk: lr = self._lev_ratio(w, vw) if lr >= 0.3: candidates.append((vw, lr * 0.5 + 0.3)) if not candidates: mapped = self._ar2en_transliterate(w) if mapped and mapped in self.vocab: result = (mapped, 0.75, True) self.correction_cache[w] = (result[0], result[1]) return result self.correction_cache[w] = (word, 0.0, False) return (word, 0.0, False) candidates.sort(key=lambda x: -x[1]) best = candidates[0] result = (best[0], best[1], best[0] != w) self.correction_cache[w] = (result[0], result[1]) return result def _ar2en_transliterate(self, word: str) -> str: mapping = { 'ا': 'a', 'أ': 'a', 'إ': 'a', 'آ': 'a', 'ء': 'a', 'ب': 'b', 'ت': 't', 'ث': 'th', 'ج': 'g', 'ح': 'h', 'خ': 'kh', 'د': 'd', 'ذ': 'th', 'ر': 'r', 'ز': 'z', 'س': 's', 'ش': 'sh', 'ص': 's', 'ض': 'd', 'ط': 't', 'ظ': 'z', 'ع': 'a', 'غ': 'gh', 'ف': 'f', 'ق': 'k', 'ك': 'k', 'ل': 'l', 'م': 'm', 'ن': 'n', 'ه': 'h', 'و': 'w', 'ي': 'y', 'ى': 'a', 'ة': 'a', 'ئ': 'a', 'ؤ': 'a', ' ': ' ', '\u200c': '', '\u200d': '', } result = '' for ch in word: if ch in mapping: result += mapping[ch] elif re.match(r'[a-zA-Z0-9]', ch): result += ch return result def correct_query(self, query: str) -> tuple: qs = query.strip() nq = normalize_text(qs) if nq in AR2EN_NORM: mapped = AR2EN_NORM[nq] if mapped: return (mapped, 0.95, True) words = qs.split() corrected = [] total_score = 0.0 was_corrected = False for w in words: cw, sc, corr = self.correct_word(w) corrected.append(cw) total_score += sc was_corrected = was_corrected or corr avg = total_score / max(len(words), 1) corrected_text = ' '.join(corrected) nc = normalize_text(corrected_text) if nc in self.normalized_vocab: corrected_text = self.normalized_vocab[nc] return (corrected_text, avg, was_corrected) class AutocompleteTrie: def __init__(self): self.children = {} self.is_end = False self.titles = [] def insert(self, title: str): node = self normalized = normalize_text(title) for ch in normalized: if ch not in node.children: node.children[ch] = AutocompleteTrie() node = node.children[ch] node.is_end = True node.titles.append(title) def _find_node(self, prefix: str): node = self normalized = normalize_text(prefix) for ch in normalized: if ch not in node.children: return None node = node.children[ch] return node def _collect(self, node, limit=10) -> list: results = [] seen = set() def dfs(n): if len(results) >= limit: return if n.is_end: for t in n.titles: if len(results) >= limit: return if t not in seen: seen.add(t) results.append(t) for ch in sorted(n.children.keys()): if len(results) < limit: dfs(n.children[ch]) dfs(node) return results def complete(self, prefix: str, limit=10) -> list: node = self._find_node(prefix) if not node: return [] return self._collect(node, limit) def search_prefix(self, prefix: str) -> bool: return self._find_node(prefix) is not None def enhanced_embed(text: str, dim: int = VECTOR_DIM) -> np.ndarray: v = np.zeros(dim, dtype=np.float64) norm = normalize_text(text) tokens = norm.split() for ti, tok in enumerate(tokens): tok_weight = 2.0 / (ti + 1) for n in range(2, 7): for i in range(len(tok) - n + 1): gram = tok[i:i+n] pos_weight = 1.0 + (2.0 / (i + 1.0)) h = abs(hash(gram)) % dim v[h] += pos_weight * tok_weight * (n ** 0.6) * 1.5 for ti, tok in enumerate(tokens): tok_weight = 4.0 / (ti + 1) h = abs(hash(tok)) % dim v[h] += tok_weight * 7.0 for i in range(len(tokens) - 1): pair = tokens[i] + ' ' + tokens[i+1] weight = 3.0 / (i + 1) h = abs(hash(pair)) % dim v[h] += weight * 5.0 for i in range(len(tokens) - 2): triple = ' '.join(tokens[i:i+3]) weight = 2.0 / (i + 1) h = abs(hash(triple)) % dim v[h] += weight * 3.0 norm_val = np.linalg.norm(v) if norm_val > 0: v /= norm_val return v.astype(np.float32) class SearchEngine: def __init__(self, index_db_path: str = INDEX_DB_PATH): self.index_db_path = index_db_path self.conn = sqlite3.connect(index_db_path) self.conn.row_factory = sqlite3.Row self._init_db() self.corrector = SpellingCorrector() self.trie = AutocompleteTrie() self.bm25 = BM25Index() self.all_vectors = np.array([]) self.all_meta = [] self.title_field = [] self.desc_field = [] self.idf = {} self.is_indexed = False def _init_db(self): c = self.conn.cursor() c.execute('''CREATE TABLE IF NOT EXISTS index_meta ( key TEXT PRIMARY KEY, value TEXT )''') c.execute('''CREATE TABLE IF NOT EXISTS search_docs ( id INTEGER PRIMARY KEY AUTOINCREMENT, archive_id INTEGER, tmdb_id INTEGER, title TEXT, media_type TEXT, year INTEGER, season INTEGER, episode INTEGER, overview TEXT, genres TEXT, cast_names TEXT, director TEXT, keywords TEXT, vector BLOB, dim INTEGER DEFAULT 1024 )''') c.execute('''CREATE TABLE IF NOT EXISTS vocab ( term TEXT PRIMARY KEY, doc_freq INTEGER, idf REAL )''') c.execute('''CREATE TABLE IF NOT EXISTS bm25_posting ( term TEXT, doc_id INTEGER, freq INTEGER, PRIMARY KEY (term, doc_id) )''') self.conn.commit() def close(self): self.conn.close() def insert_or_update_doc(self, archive_id: int, tmdb_id: int, title: str, media_type: str, year: int, season: int, episode: int, overview: str, genres: str, cast_names: str, director: str, keywords: str, vector: np.ndarray): c = self.conn.cursor() vec_bytes = vector.tobytes() existing = c.execute( 'SELECT id FROM search_docs WHERE archive_id=? AND tmdb_id=?', (archive_id, tmdb_id) ).fetchone() if existing: c.execute('''UPDATE search_docs SET title=?, media_type=?, year=?, season=?, episode=?, overview=?, genres=?, cast_names=?, director=?, keywords=?, vector=?, dim=? WHERE id=?''', (title, media_type, year, season, episode, overview, genres, cast_names, director, keywords, vec_bytes, VECTOR_DIM, existing[0])) return existing[0] else: c.execute('''INSERT INTO search_docs (archive_id, tmdb_id, title, media_type, year, season, episode, overview, genres, cast_names, director, keywords, vector, dim) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?)''', (archive_id, tmdb_id, title, media_type, year, season, episode, overview, genres, cast_names, director, keywords, vec_bytes, VECTOR_DIM)) self.conn.commit() return c.lastrowid def rebuild_from_archive(self, archive_db_path: str, verbose=True): if verbose: print(" 🔄 Rebuilding search index from archive DB...") conn_ar = sqlite3.connect(archive_db_path) conn_ar.row_factory = sqlite3.Row rows = conn_ar.execute(''' SELECT a.id, a.title, a.year, a.media_type, a.tmdb_id, a.season, a.episode, a.highlights, a.full_metadata FROM archives a ORDER BY a.id ''').fetchall() conn_ar.close() entries = [] for r in rows: fm = json.loads(r['full_metadata']) if r['full_metadata'] else {} hl = json.loads(r['highlights']) if r['highlights'] else {} details = fm.get('details', {}) if isinstance(fm, dict) else {} credits = fm.get('credits', {}) if isinstance(fm, dict) else {} overview = details.get('overview', '') or hl.get('overview', '') genres_list = details.get('genres', []) or [] genres_str = ', '.join(g['name'] for g in genres_list if isinstance(g, dict)) cast_list = credits.get('cast', []) if isinstance(credits, dict) else [] cast_names = ' '.join(c.get('name', '') for c in cast_list[:20]) crew_list = credits.get('crew', []) if isinstance(credits, dict) else [] director = '' for c in crew_list: if isinstance(c, dict) and c.get('job') == 'Director': director = c.get('name', '') break kw_data = fm.get('keywords', {}) if isinstance(kw_data, dict): kw_list = kw_data.get('keywords', []) or kw_data.get('results', []) else: kw_list = [] kw_str = ' '.join(k.get('name', '') for k in kw_list if isinstance(k, dict)) entries.append({ 'id': r['id'], 'title': r['title'] or '', 'year': r['year'], 'media_type': r['media_type'], 'tmdb_id': r['tmdb_id'], 'season': r['season'], 'episode': r['episode'], 'overview': overview or '', 'genres': genres_str or '', 'cast_names': cast_names or '', 'director': director or '', 'keywords': kw_str or '', }) titles = [e['title'] for e in entries if e['title']] self.corrector = SpellingCorrector() self.trie = AutocompleteTrie() self.corrector.add_titles(titles) extra = ['فيلم', 'مسلسل', 'movie', 'series', 'season', 'موسم', 'حلقة', 'episode', '1080p', '720p', '2160p', '4k', 'bluray', 'webrip', 'hdtv', 'x264', 'x265'] self.corrector.add_words(extra) for t in titles: if t: self.trie.insert(t) vecs = [] metas = [] self.title_field = [] self.desc_field = [] for e in entries: title = e['title'] desc = e['overview'] genres = e['genres'] cast = e['cast_names'] director = e['director'] kw = e['keywords'] combined = title if desc: combined += ' ' + desc if genres: combined += ' ' + genres if cast: combined += ' ' + cast if director: combined += ' ' + director if kw: combined += ' ' + kw vec = enhanced_embed(combined) vecs.append(vec) metas.append(e) self.title_field.append(tokenize(title)) desc_text = f"{desc} {genres} {cast} {director} {kw}" self.desc_field.append(tokenize(desc_text)) self.all_vectors = np.array(vecs) if vecs else np.array([]) self.all_meta = metas self.titles_cache = titles all_docs = [t + d for t, d in zip(self.title_field, self.desc_field)] self.idf = compute_tfidf(all_docs) self.bm25.build(all_docs, self.idf) self.is_indexed = True c = self.conn.cursor() c.execute("DELETE FROM search_docs") c.execute("DELETE FROM vocab") c.execute("DELETE FROM bm25_posting") for i, e in enumerate(entries): self.insert_or_update_doc( e['id'], e['tmdb_id'], e['title'], e['media_type'], e['year'], e['season'], e['episode'], e['overview'], e['genres'], e['cast_names'], e['director'], e['keywords'], vecs[i] ) for term, idf_val in self.idf.items(): df = sum(1 for doc in all_docs if term in doc) c.execute("INSERT OR REPLACE INTO vocab VALUES (?,?,?)", (term, df, idf_val)) for term, posting in self.bm25.term_doc_freq.items(): for doc_id, freq in posting.items(): c.execute("INSERT OR REPLACE INTO bm25_posting VALUES (?,?,?)", (term, doc_id, freq)) c.execute("INSERT OR REPLACE INTO index_meta VALUES ('entry_count', ?)", (str(len(entries)),)) c.execute("INSERT OR REPLACE INTO index_meta VALUES ('vector_dim', ?)", (str(VECTOR_DIM),)) self.conn.commit() if verbose: print(f" ✅ Indexed {len(entries)} entries, " f"vocab={len(self.idf)} terms, " f"dim={VECTOR_DIM}") def autocomplete(self, prefix: str, limit: int = 10) -> list: return self.trie.complete(prefix, limit) def correct_query(self, query: str) -> tuple: return self.corrector.correct_query(query) def search_by_vector(self, query: str, top_k: int = 30) -> list: if not self.is_indexed or len(self.all_vectors) == 0: return [] qv = enhanced_embed(query) sims = np.dot(self.all_vectors, qv) top = np.argsort(sims)[::-1][:top_k] results = [] seen_ids = set() for idx in top: if sims[idx] > 0.0: m = self.all_meta[idx] dedup_key = (m['tmdb_id'], m.get('season'), m.get('episode')) if dedup_key not in seen_ids: seen_ids.add(dedup_key) results.append({**m, 'score': float(sims[idx])}) return results def search_by_keyword(self, query: str, top_k: int = 30) -> list: if not self.is_indexed: return [] qtokens = tokenize(query) if not qtokens: return [] bm25_results = self.bm25.search(qtokens, top_k) results = [] seen_ids = set() for idx, score in bm25_results: m = self.all_meta[idx] dedup_key = (m['tmdb_id'], m.get('season'), m.get('episode')) if dedup_key not in seen_ids: seen_ids.add(dedup_key) results.append({**m, 'score': float(score), 'match_type': 'keyword'}) return results def _expand_arabic_query(self, query: str) -> str: tokens = tokenize(query) ar_count = sum(1 for t in tokens if re.search(r'[\u0600-\u06FF]', t)) if ar_count == 0 or len(tokens) <= 2: return query translated = [] for tok in tokens: if re.search(r'[\u0600-\u06FF]', tok): for ar_word, en_words in sorted(AR_DESC_EN.items(), key=lambda x: -len(x[0])): if ar_word in tok: translated.append(en_words) break else: translated.append(tok) else: translated.append(tok) return ' '.join(translated) def search_hybrid(self, query: str, top_k: int = 30, vector_weight: float = 0.6) -> list: extended_query = self._expand_arabic_query(query) if extended_query != query: vec_results = self.search_by_vector(extended_query, top_k * 2) kw_results = self.search_by_keyword(extended_query, top_k * 2) else: vec_results = self.search_by_vector(query, top_k * 2) kw_results = self.search_by_keyword(query, top_k * 2) def norm_scores(results_list): if not results_list: return results_list scores = [r['score'] for r in results_list] mn, mx = min(scores), max(scores) rng = mx - mn if mx > mn else 1.0 for r in results_list: r['score'] = (r['score'] - mn) / rng return results_list vec_results = norm_scores(vec_results) kw_results = norm_scores(kw_results) tok_count = len(tokenize(query)) vw = 0.75 if tok_count <= 2 else vector_weight seen = {} for r in vec_results: key = (r['tmdb_id'], r.get('season'), r.get('episode')) seen[key] = r r['vec_score'] = r['score'] r['kw_score'] = 0.0 for r in kw_results: key = (r['tmdb_id'], r.get('season'), r.get('episode')) if key in seen: seen[key]['kw_score'] = r['score'] seen[key]['score'] = (seen[key]['vec_score'] * vw + r['score'] * (1 - vw)) else: r['vec_score'] = 0.0 r['kw_score'] = r['score'] r['score'] = r['score'] * (1 - vw) seen[key] = r results = sorted(seen.values(), key=lambda x: -x['score'])[:top_k] return results def search_with_suggestions(self, query: str, top_k: int = 30) -> dict: if not self.is_indexed: return {'query': query, 'corrected_query': query, 'was_corrected': False, 'results': [], 'suggestions': [], 'total': 0} corrected, conf, was_corrected = self.corrector.correct_query(query) results = self.search_hybrid(corrected, top_k) suggestions = [] if was_corrected and corrected != query and conf >= 0.5: suggestions.append({ 'type': 'spelling', 'title': corrected, 'original': query, 'confidence': conf }) if not results or results[0]['score'] < 0.15: ac = self.autocomplete(query, 5) for t in ac: if t not in [r['title'] for r in results]: suggestions.append({'type': 'autocomplete', 'title': t}) ac_results = self.autocomplete(query, 5) for t in ac_results: if t not in [r['title'] for r in results]: has = any((s.get('title') or s.get('corrected', '')) == t for s in suggestions) if not has: suggestions.append({'type': 'autocomplete', 'title': t}) return { 'query': query, 'corrected_query': corrected if was_corrected else query, 'was_corrected': was_corrected, 'results': results, 'suggestions': suggestions, 'total': len(results) } def interactive_search(self): print(f"\n{'='*60}") print(f" 🔍 PopCorn Search Engine v2 (1024-dim + BM25)") print(f" {'='*60}") print(f" Type your query, or 'q' to quit, '!index' to rebuild") print(f" {'='*60}") if not self.is_indexed: self.rebuild_from_archive( os.path.join(os.path.dirname(self.index_db_path), 'archive_data.db')) while True: try: query = input(f"\n Search > ").strip() except (EOFError, KeyboardInterrupt): break if not query: continue if query.lower() in ('q', 'quit', 'exit'): break if query.lower() == '!index': self.rebuild_from_archive( os.path.join(os.path.dirname(self.index_db_path), 'archive_data.db')) continue result = self.search_with_suggestions(query) self._print_results(result, query) def _print_results(self, result: dict, query: str): r = result rlist = r['results'][:10] if r['was_corrected']: c = r['corrected_query'] if c != query: print(f"\n \033[33m Showing results for \033[1m{c}\033[0m\033[33m" f" (searched for: \033[1m{r['query']}\033[0m\033[33m)\033[0m") for s in r['suggestions']: if s['type'] == 'autocomplete': print(f" \033[36m Did you mean: \033[1m{s['title']}\033[0m") if not rlist: print(f"\n \033[33m No results for '{query}'.\033[0m") if r.get('suggestions'): print(f" \033[36m Try: {', '.join(s['title'] for s in r['suggestions'])}\033[0m") return print(f"\n \033[32m \033[1m{result['total']}\033[0m\033[32m results\033[0m " f"(\033[33m{r['results'][0]['score']:.4f}\033[0m - " f"\033[33m{r['results'][-1]['score']:.4f}\033[0m)\n") for i, res in enumerate(rlist): mt = 'M' if res['media_type'] == 'movie' else 'T' si = f" S{res['season']}E{res['episode']}" if res.get('season') else '' kw = f" [{res.get('match_type', 'hybrid')}]" if res.get('match_type') else '' print(f" \033[1m{i+1}. {mt} {res['title']} ({res['year'] or '?'})\033[0m" f"{si}{kw}") print(f" ID={res['tmdb_id']} score=\033[33m{res['score']:.4f}\033[0m") if res.get('overview'): print(f" \033[90m{res['overview'][:150]}...\033[0m") if res.get('genres'): print(f" {res['genres']}") if res.get('director'): print(f" {res['director']}") if res.get('cast_names'): cn = res['cast_names'][:80] print(f" {cn}") print() def get_search_snippet(self, query: str) -> str: result = self.search_with_suggestions(query, top_k=3) lines = [] if result['was_corrected']: lines.append(f"Showing results for: {result['corrected_query']} (searched: {result['query']})") for s in result['suggestions']: if s['type'] == 'autocomplete': lines.append(f"Did you mean: {s['title']}") for r in result['results'][:5]: mt = 'M' if r['media_type'] == 'movie' else 'T' lines.append(f"{mt} {r['title']} ({r['year'] or '?'}) [{r['score']:.4f}]") return '\n'.join(lines) def store_crew_images(conn, archive_id: int, tmdb_id: int, mt: str, full: dict): c = conn.cursor() credits = full.get('credits', {}) or {} crew = credits.get('crew', []) if isinstance(credits, dict) else [] seen = set() for person in crew: role = person.get('job', '') if role in ('Director', 'Writer', 'Producer', 'Screenplay', 'Story', 'Executive Producer'): pid = person.get('id', 0) if pid in seen or not pid: continue seen.add(pid) pp = person.get('profile_path') if pp: from urllib.request import Request, urlopen try: req = Request(f"https://image.tmdb.org/t/p/original{pp}", headers={'User-Agent': 'PopCorn/1.0'}) data = urlopen(req, timeout=30).read() name = f"profile/{pid}_{0}.jpg" c.execute('INSERT OR IGNORE INTO media_cache ' '(archive_id, tmdb_id, media_subtype, local_path, source_url, data, person_id) ' 'VALUES (?,?,?,?,?,?,?)', (archive_id, tmdb_id, 'crew_profile', name, f"https://image.tmdb.org/t/p/original{pp}", data, pid)) except Exception: pass conn.commit() def generate_training_data(db_path: str, output_path: str = None): conn = sqlite3.connect(db_path) c = conn.cursor() rows = c.execute('SELECT id, title, year, tmdb_id, media_type, highlights FROM archives').fetchall() samples = [] for r in rows: entry = {'id': r[0], 'title': r[1], 'year': r[2], 'tmdb_id': r[3], 'media_type': r[4]} h = json.loads(r[5]) if r[5] else {} entry['overview'] = h.get('overview', '') entry['genres'] = h.get('genres', '') entry['director'] = h.get('director', '') entry['cast'] = [c.get('name', '') for c in h.get('cast', [])[:5]] samples.append(entry) alt_titles = [] title = entry['title'] or '' if title: at = title.lower().strip() alt_titles.extend([ at.replace(' ', ''), re.sub(r'[aeiou]', '', at), at.replace('the ', '').replace('The ', ''), ]) if re.search(r'[a-zA-Z]', at): try: ar_trans = at.translate(str.maketrans('abcdefghijklmnopqrstuvwxyz', 'ابجد هوز حطي كلمن سعفص قرشت ثخذ ضظغ')) alt_titles.append(ar_trans) except: pass samples[-1]['alt_titles'] = alt_titles conn.close() if output_path: with open(output_path, 'w', encoding='utf-8') as f: json.dump(samples, f, ensure_ascii=False, indent=2) print(f" Training data saved: {output_path} ({len(samples)} entries)") return samples