Spaces:
Running
Running
| import re | |
| import sqlite3 | |
| import os | |
| import sys | |
| from pathlib import Path | |
| # Add app to path to import services/db | |
| sys.path.append(str(Path(__file__).parent.parent.parent)) | |
| from app.database.sqlite_db import get_db | |
| def extract_references(): | |
| db = get_db() | |
| print(f"Starting reference indexing on {db.db_path}...") | |
| # We use disk connection for writes | |
| with db.get_disk_connection() as conn: | |
| cursor = conn.cursor() | |
| # 1. Get all pages | |
| cursor.execute(""" | |
| SELECT p.id, v.volume_number, p.page_number, p.content_html | |
| FROM pages p | |
| JOIN volumes v ON p.volume_id = v.id | |
| WHERE p.content_html IS NOT NULL | |
| """) | |
| pages = cursor.fetchall() | |
| total = len(pages) | |
| print(f"Processing {total} pages...") | |
| indexed_count = 0 | |
| for i, page in enumerate(pages): | |
| vol_num = page["volume_number"] | |
| page_num = page["page_number"] | |
| raw_html = page["content_html"] | |
| # Extract footnotes (@ markers) | |
| # Strip line numbers first | |
| fn_html = re.sub(r'<span\s+class="LineNumber">.*?</span>', '', raw_html) | |
| lines = fn_html.split('\n') | |
| current_fn = None | |
| for line in lines: | |
| line_strip = line.strip() | |
| if line_strip.startswith('@'): | |
| line_plain = re.sub(r'<[^>]+>', '', line_strip) | |
| # Match @ marker content | |
| m = re.match(r'^@\s*([(\[]?[\u0E50-\u0E59\d]+[)\]-]?)\s*(.*)', line_plain) | |
| if m: | |
| marker = m.group(1).strip() | |
| text = m.group(2).strip() | |
| clean_marker = re.sub(r'[()\[\]-]', '', marker).strip() | |
| if clean_marker and not re.search(r'เชิงอรรถ', marker): | |
| # Save current if exists | |
| if current_fn: | |
| save_ref(cursor, vol_num, page_num, current_fn["id"], "footnote", current_fn["content"]) | |
| indexed_count += 1 | |
| current_fn = {"id": clean_marker, "content": text} | |
| elif current_fn: | |
| # Continuation line starting with @ | |
| current_fn["content"] += " " + line_plain.lstrip('@').strip() | |
| elif current_fn: | |
| # Non-@ line ends the footnote block | |
| save_ref(cursor, vol_num, page_num, current_fn["id"], "footnote", current_fn["content"]) | |
| indexed_count += 1 | |
| current_fn = None | |
| if current_fn: | |
| save_ref(cursor, vol_num, page_num, current_fn["id"], "footnote", current_fn["content"]) | |
| indexed_count += 1 | |
| # Extract abbreviations (ย่อ) - this is harder as they are often just (ย่อ) in text | |
| # For now, we can look for specific (ย่อ) patterns that might have been manually defined | |
| # or common ones we want to pre-cache. | |
| # In the current DB, most (ย่อ) are in-line. | |
| # If there are any @(ย่อ) or similar, we catch them above. | |
| if i % 1000 == 0: | |
| print(f"Processed {i}/{total} pages... Indexed {indexed_count} markers") | |
| conn.commit() | |
| conn.commit() | |
| print(f"Finished! Total indexed: {indexed_count}") | |
| def save_ref(cursor, vol, page, marker_id, ref_type, content): | |
| try: | |
| cursor.execute(""" | |
| INSERT OR REPLACE INTO reference_markers (volume_num, page_num, marker_id, type, content) | |
| VALUES (?, ?, ?, ?, ?) | |
| """, (vol, page, marker_id, ref_type, content.strip())) | |
| except Exception as e: | |
| print(f"Error saving ref {vol}:{page} {marker_id}: {e}") | |
| if __name__ == "__main__": | |
| extract_references() | |