Download utils/wiki_pull_phasma.py from gaaaaaaaaaaa/multimodal-reasoning: direct link, hf CLI and curl.
- Browser
- Download file 9.47 kB
-
https://huggingface.co/gaaaaaaaaaaa/multimodal-reasoning/resolve/main/utils/wiki_pull_phasma.py
- Command line
-
hf download hf://gaaaaaaaaaaa/multimodal-reasoning/utils/wiki_pull_phasma.py
-
curl -L -o wiki_pull_phasma.py https://huggingface.co/gaaaaaaaaaaa/multimodal-reasoning/resolve/main/utils/wiki_pull_phasma.py
9.47 kB
| # scripts/pull_phasmophobia_wiki.py | |
| from __future__ import annotations | |
| import json | |
| import re | |
| import time | |
| from pathlib import Path | |
| from urllib.parse import quote | |
| import requests | |
| from bs4 import BeautifulSoup | |
| from tqdm.auto import tqdm | |
| API_URL = "https://phasmophobia.fandom.com/api.php" | |
| BASE_URL = "https://phasmophobia.fandom.com/wiki" | |
| OUT_DIR = Path("Resources/Documents") | |
| OUT_DOCS = OUT_DIR / "phasmophobia_wiki_docs.jsonl" | |
| OUT_CHUNKS = OUT_DIR / "phasmophobia_wiki_chunks.jsonl" | |
| SESSION = requests.Session() | |
| SESSION.headers.update({ | |
| "User-Agent": "MultimodalReasoningBot/0.1 clean-document-pull" | |
| }) | |
| REMOVE_SELECTORS = [ | |
| "script", | |
| "style", | |
| "noscript", | |
| "figure", | |
| "img", | |
| "audio", | |
| "video", | |
| "iframe", | |
| "sup.reference", | |
| ".reference", | |
| ".references", | |
| ".reflist", | |
| ".mw-editsection", | |
| ".mw-empty-elt", | |
| ".toc", | |
| "#toc", | |
| ".portable-infobox", | |
| ".infobox", | |
| ".navbox", | |
| ".metadata", | |
| ".ambox", | |
| ".catlinks", | |
| ".printfooter", | |
| ".noprint", | |
| ".thumb", | |
| ".gallery", | |
| ".wikia-gallery", | |
| ".pi-image", | |
| ".pi-data", | |
| ".pi-header", | |
| ".pi-title", | |
| ".page-header", | |
| ".page-footer", | |
| ".license-description", | |
| ".WikiaArticleFooter", | |
| ".rail-module", | |
| ] | |
| SKIP_TITLE_PREFIXES = ( | |
| "User:", | |
| "User talk:", | |
| "Talk:", | |
| "File:", | |
| "Template:", | |
| "Category:", | |
| "Help:", | |
| "Forum:", | |
| "Blog:", | |
| "MediaWiki:", | |
| "Module:", | |
| ) | |
| SKIP_EXACT_TITLES = { | |
| "Main Page", | |
| } | |
| BAD_LINE_PATTERNS = [ | |
| r"^advertisement$", | |
| r"^contents$", | |
| r"^categories$", | |
| r"^references$", | |
| r"^gallery$", | |
| r"^trivia$", | |
| r"^see also$", | |
| r"^external links$", | |
| r"^community content is available", | |
| r"^fandom apps", | |
| r"^take your favorite fandoms", | |
| r"^explore properties", | |
| r"^view mobile site", | |
| r"^follow on", | |
| r"^sign in", | |
| r"^create a free account", | |
| ] | |
| def api_get(params: dict) -> dict: | |
| params = { | |
| "format": "json", | |
| "formatversion": "2", | |
| **params, | |
| } | |
| for attempt in range(5): | |
| r = SESSION.get(API_URL, params=params, timeout=30) | |
| if r.status_code == 429: | |
| time.sleep(2 + attempt) | |
| continue | |
| r.raise_for_status() | |
| return r.json() | |
| raise RuntimeError(f"API failed after retries: {params}") | |
| def get_all_pages() -> list[dict]: | |
| pages = [] | |
| apcontinue = None | |
| while True: | |
| params = { | |
| "action": "query", | |
| "list": "allpages", | |
| "apnamespace": 0, | |
| "aplimit": "max", | |
| "apfilterredir": "nonredirects", | |
| } | |
| if apcontinue: | |
| params["apcontinue"] = apcontinue | |
| data = api_get(params) | |
| pages.extend(data.get("query", {}).get("allpages", [])) | |
| cont = data.get("continue", {}) | |
| apcontinue = cont.get("apcontinue") | |
| if not apcontinue: | |
| break | |
| clean = [] | |
| for page in pages: | |
| title = page["title"].strip() | |
| if title in SKIP_EXACT_TITLES: | |
| continue | |
| if title.startswith(SKIP_TITLE_PREFIXES): | |
| continue | |
| clean.append(page) | |
| return clean | |
| def get_parsed_html(pageid: int) -> str | None: | |
| data = api_get({ | |
| "action": "parse", | |
| "pageid": pageid, | |
| "prop": "text|displaytitle", | |
| "redirects": "1", | |
| "disableeditsection": "1", | |
| "disabletoc": "1", | |
| }) | |
| parsed = data.get("parse") | |
| if not parsed: | |
| return None | |
| text = parsed.get("text") | |
| if isinstance(text, dict): | |
| return text.get("*") | |
| return text | |
| def clean_line(line: str) -> str: | |
| line = re.sub(r"\[\s*edit\s*\]", "", line, flags=re.I) | |
| line = re.sub(r"\[\d+\]", "", line) | |
| line = re.sub(r"\s+", " ", line) | |
| return line.strip() | |
| def is_bad_line(line: str) -> bool: | |
| if not line: | |
| return True | |
| low = line.lower().strip() | |
| if len(low) <= 1: | |
| return True | |
| for pattern in BAD_LINE_PATTERNS: | |
| if re.search(pattern, low): | |
| return True | |
| if low.startswith("http://") or low.startswith("https://"): | |
| return True | |
| if "retrieved from" in low: | |
| return True | |
| if "fandom.com" in low and len(low.split()) < 16: | |
| return True | |
| return False | |
| def html_to_clean_text(html: str, keep_tables: bool = False) -> str: | |
| soup = BeautifulSoup(html, "lxml") | |
| root = soup.select_one(".mw-parser-output") | |
| if root is None: | |
| root = soup | |
| for selector in REMOVE_SELECTORS: | |
| for tag in root.select(selector): | |
| tag.decompose() | |
| if not keep_tables: | |
| for tag in root.find_all("table"): | |
| tag.decompose() | |
| blocks = [] | |
| for tag in root.find_all(["h2", "h3", "h4", "p", "li"]): | |
| text = clean_line(tag.get_text(" ", strip=True)) | |
| if is_bad_line(text): | |
| continue | |
| if tag.name in {"h2", "h3", "h4"}: | |
| # Keep useful section boundaries, but not table-of-contents garbage. | |
| blocks.append(f"\n## {text}\n") | |
| else: | |
| blocks.append(text) | |
| text = "\n".join(blocks) | |
| text = re.sub(r"\n{3,}", "\n\n", text) | |
| text = re.sub(r"[ \t]{2,}", " ", text) | |
| return text.strip() | |
| def title_to_url(title: str) -> str: | |
| return f"{BASE_URL}/{quote(title.replace(' ', '_'))}" | |
| def pull_docs(limit: int | None = None, test_pages: list[str] | None = None) -> list[dict]: | |
| OUT_DIR.mkdir(parents=True, exist_ok=True) | |
| if test_pages: | |
| all_pages = get_all_pages() | |
| wanted = {x.lower() for x in test_pages} | |
| pages = [p for p in all_pages if p["title"].lower() in wanted] | |
| else: | |
| pages = get_all_pages() | |
| if limit is not None: | |
| pages = pages[:limit] | |
| print(f"[Pull] Pages selected: {len(pages)}") | |
| records = [] | |
| for page in tqdm(pages, desc="Pulling Phasmophobia wiki"): | |
| pageid = int(page["pageid"]) | |
| title = page["title"].strip() | |
| try: | |
| html = get_parsed_html(pageid) | |
| if not html: | |
| continue | |
| text = html_to_clean_text(html, keep_tables=False) | |
| # Skip pages that are only nav/category/etc. | |
| if len(text.split()) < 40: | |
| continue | |
| records.append({ | |
| "source": "phasmophobia.fandom.com", | |
| "pageid": pageid, | |
| "title": title, | |
| "url": title_to_url(title), | |
| "text": text, | |
| }) | |
| time.sleep(0.05) | |
| except Exception as e: | |
| print(f"[WARN] Failed {title}: {e}") | |
| return records | |
| def chunk_text(text: str, chunk_words: int = 256, overlap: int = 48) -> list[str]: | |
| words = text.split() | |
| if len(words) <= chunk_words: | |
| return [" ".join(words)] | |
| chunks = [] | |
| step = max(1, chunk_words - overlap) | |
| for start in range(0, len(words), step): | |
| piece = words[start:start + chunk_words] | |
| if len(piece) < 40: | |
| continue | |
| chunks.append(" ".join(piece)) | |
| return chunks | |
| def save_docs(records: list[dict]) -> None: | |
| OUT_DIR.mkdir(parents=True, exist_ok=True) | |
| with OUT_DOCS.open("w", encoding="utf-8") as f: | |
| for record in records: | |
| f.write(json.dumps(record, ensure_ascii=False) + "\n") | |
| print(f"[Save] Docs: {OUT_DOCS} ({len(records)} records)") | |
| def save_chunks(records: list[dict], chunk_words: int = 256, overlap: int = 48) -> None: | |
| count = 0 | |
| with OUT_CHUNKS.open("w", encoding="utf-8") as f: | |
| for record in records: | |
| chunks = chunk_text( | |
| record["text"], | |
| chunk_words=chunk_words, | |
| overlap=overlap, | |
| ) | |
| for chunk_id, chunk in enumerate(chunks): | |
| f.write(json.dumps({ | |
| "source": record["source"], | |
| "title": record["title"], | |
| "url": record["url"], | |
| "pageid": record["pageid"], | |
| "chunk_id": chunk_id, | |
| "text": chunk, | |
| }, ensure_ascii=False) + "\n") | |
| count += 1 | |
| print(f"[Save] Chunks: {OUT_CHUNKS} ({count} chunks)") | |
| def preview(records: list[dict], max_docs: int = 5, chars: int = 1200) -> None: | |
| for record in records[:max_docs]: | |
| print("=" * 100) | |
| print(record["title"]) | |
| print(record["url"]) | |
| print("-" * 100) | |
| print(record["text"][:chars]) | |
| def quality_check(records: list[dict]) -> None: | |
| bad_markers = [ | |
| "Advertisement", | |
| "Fandom Apps", | |
| "Take your favorite fandoms", | |
| "View Mobile Site", | |
| "Create a Free Account", | |
| "Sign In", | |
| "Retrieved from", | |
| ] | |
| print("\n[Quality Check]") | |
| for marker in bad_markers: | |
| hits = [ | |
| r["title"] | |
| for r in records | |
| if marker.lower() in r["text"].lower() | |
| ] | |
| print(f"{marker!r}: {len(hits)} hits") | |
| if hits[:5]: | |
| print(" sample:", hits[:5]) | |
| def main(): | |
| # Test first against known pages I checked manually: | |
| # Phasmophobia, Ghost, Map | |
| test = False | |
| if test: | |
| records = pull_docs(test_pages=["Phasmophobia", "Ghost", "Map"]) | |
| else: | |
| records = pull_docs() | |
| save_docs(records) | |
| save_chunks(records, chunk_words=256, overlap=48) | |
| quality_check(records) | |
| preview(records, max_docs=3) | |
| if __name__ == "__main__": | |
| main() |