# scripts/pull_phasmophobia_wiki.py from __future__ import annotations import json import re import time from pathlib import Path from urllib.parse import quote import requests from bs4 import BeautifulSoup from tqdm.auto import tqdm API_URL = "https://phasmophobia.fandom.com/api.php" BASE_URL = "https://phasmophobia.fandom.com/wiki" OUT_DIR = Path("Resources/Documents") OUT_DOCS = OUT_DIR / "phasmophobia_wiki_docs.jsonl" OUT_CHUNKS = OUT_DIR / "phasmophobia_wiki_chunks.jsonl" SESSION = requests.Session() SESSION.headers.update({ "User-Agent": "MultimodalReasoningBot/0.1 clean-document-pull" }) REMOVE_SELECTORS = [ "script", "style", "noscript", "figure", "img", "audio", "video", "iframe", "sup.reference", ".reference", ".references", ".reflist", ".mw-editsection", ".mw-empty-elt", ".toc", "#toc", ".portable-infobox", ".infobox", ".navbox", ".metadata", ".ambox", ".catlinks", ".printfooter", ".noprint", ".thumb", ".gallery", ".wikia-gallery", ".pi-image", ".pi-data", ".pi-header", ".pi-title", ".page-header", ".page-footer", ".license-description", ".WikiaArticleFooter", ".rail-module", ] SKIP_TITLE_PREFIXES = ( "User:", "User talk:", "Talk:", "File:", "Template:", "Category:", "Help:", "Forum:", "Blog:", "MediaWiki:", "Module:", ) SKIP_EXACT_TITLES = { "Main Page", } BAD_LINE_PATTERNS = [ r"^advertisement$", r"^contents$", r"^categories$", r"^references$", r"^gallery$", r"^trivia$", r"^see also$", r"^external links$", r"^community content is available", r"^fandom apps", r"^take your favorite fandoms", r"^explore properties", r"^view mobile site", r"^follow on", r"^sign in", r"^create a free account", ] def api_get(params: dict) -> dict: params = { "format": "json", "formatversion": "2", **params, } for attempt in range(5): r = SESSION.get(API_URL, params=params, timeout=30) if r.status_code == 429: time.sleep(2 + attempt) continue r.raise_for_status() return r.json() raise RuntimeError(f"API failed after retries: {params}") def get_all_pages() -> list[dict]: pages = [] apcontinue = None while True: params = { "action": "query", "list": "allpages", "apnamespace": 0, "aplimit": "max", "apfilterredir": "nonredirects", } if apcontinue: params["apcontinue"] = apcontinue data = api_get(params) pages.extend(data.get("query", {}).get("allpages", [])) cont = data.get("continue", {}) apcontinue = cont.get("apcontinue") if not apcontinue: break clean = [] for page in pages: title = page["title"].strip() if title in SKIP_EXACT_TITLES: continue if title.startswith(SKIP_TITLE_PREFIXES): continue clean.append(page) return clean def get_parsed_html(pageid: int) -> str | None: data = api_get({ "action": "parse", "pageid": pageid, "prop": "text|displaytitle", "redirects": "1", "disableeditsection": "1", "disabletoc": "1", }) parsed = data.get("parse") if not parsed: return None text = parsed.get("text") if isinstance(text, dict): return text.get("*") return text def clean_line(line: str) -> str: line = re.sub(r"\[\s*edit\s*\]", "", line, flags=re.I) line = re.sub(r"\[\d+\]", "", line) line = re.sub(r"\s+", " ", line) return line.strip() def is_bad_line(line: str) -> bool: if not line: return True low = line.lower().strip() if len(low) <= 1: return True for pattern in BAD_LINE_PATTERNS: if re.search(pattern, low): return True if low.startswith("http://") or low.startswith("https://"): return True if "retrieved from" in low: return True if "fandom.com" in low and len(low.split()) < 16: return True return False def html_to_clean_text(html: str, keep_tables: bool = False) -> str: soup = BeautifulSoup(html, "lxml") root = soup.select_one(".mw-parser-output") if root is None: root = soup for selector in REMOVE_SELECTORS: for tag in root.select(selector): tag.decompose() if not keep_tables: for tag in root.find_all("table"): tag.decompose() blocks = [] for tag in root.find_all(["h2", "h3", "h4", "p", "li"]): text = clean_line(tag.get_text(" ", strip=True)) if is_bad_line(text): continue if tag.name in {"h2", "h3", "h4"}: # Keep useful section boundaries, but not table-of-contents garbage. blocks.append(f"\n## {text}\n") else: blocks.append(text) text = "\n".join(blocks) text = re.sub(r"\n{3,}", "\n\n", text) text = re.sub(r"[ \t]{2,}", " ", text) return text.strip() def title_to_url(title: str) -> str: return f"{BASE_URL}/{quote(title.replace(' ', '_'))}" def pull_docs(limit: int | None = None, test_pages: list[str] | None = None) -> list[dict]: OUT_DIR.mkdir(parents=True, exist_ok=True) if test_pages: all_pages = get_all_pages() wanted = {x.lower() for x in test_pages} pages = [p for p in all_pages if p["title"].lower() in wanted] else: pages = get_all_pages() if limit is not None: pages = pages[:limit] print(f"[Pull] Pages selected: {len(pages)}") records = [] for page in tqdm(pages, desc="Pulling Phasmophobia wiki"): pageid = int(page["pageid"]) title = page["title"].strip() try: html = get_parsed_html(pageid) if not html: continue text = html_to_clean_text(html, keep_tables=False) # Skip pages that are only nav/category/etc. if len(text.split()) < 40: continue records.append({ "source": "phasmophobia.fandom.com", "pageid": pageid, "title": title, "url": title_to_url(title), "text": text, }) time.sleep(0.05) except Exception as e: print(f"[WARN] Failed {title}: {e}") return records def chunk_text(text: str, chunk_words: int = 256, overlap: int = 48) -> list[str]: words = text.split() if len(words) <= chunk_words: return [" ".join(words)] chunks = [] step = max(1, chunk_words - overlap) for start in range(0, len(words), step): piece = words[start:start + chunk_words] if len(piece) < 40: continue chunks.append(" ".join(piece)) return chunks def save_docs(records: list[dict]) -> None: OUT_DIR.mkdir(parents=True, exist_ok=True) with OUT_DOCS.open("w", encoding="utf-8") as f: for record in records: f.write(json.dumps(record, ensure_ascii=False) + "\n") print(f"[Save] Docs: {OUT_DOCS} ({len(records)} records)") def save_chunks(records: list[dict], chunk_words: int = 256, overlap: int = 48) -> None: count = 0 with OUT_CHUNKS.open("w", encoding="utf-8") as f: for record in records: chunks = chunk_text( record["text"], chunk_words=chunk_words, overlap=overlap, ) for chunk_id, chunk in enumerate(chunks): f.write(json.dumps({ "source": record["source"], "title": record["title"], "url": record["url"], "pageid": record["pageid"], "chunk_id": chunk_id, "text": chunk, }, ensure_ascii=False) + "\n") count += 1 print(f"[Save] Chunks: {OUT_CHUNKS} ({count} chunks)") def preview(records: list[dict], max_docs: int = 5, chars: int = 1200) -> None: for record in records[:max_docs]: print("=" * 100) print(record["title"]) print(record["url"]) print("-" * 100) print(record["text"][:chars]) def quality_check(records: list[dict]) -> None: bad_markers = [ "Advertisement", "Fandom Apps", "Take your favorite fandoms", "View Mobile Site", "Create a Free Account", "Sign In", "Retrieved from", ] print("\n[Quality Check]") for marker in bad_markers: hits = [ r["title"] for r in records if marker.lower() in r["text"].lower() ] print(f"{marker!r}: {len(hits)} hits") if hits[:5]: print(" sample:", hits[:5]) def main(): # Test first against known pages I checked manually: # Phasmophobia, Ghost, Map test = False if test: records = pull_docs(test_pages=["Phasmophobia", "Ghost", "Map"]) else: records = pull_docs() save_docs(records) save_chunks(records, chunk_words=256, overlap=48) quality_check(records) preview(records, max_docs=3) if __name__ == "__main__": main()