multimodal-reasoning / utils /wiki_pull_phasma.py
gaaaaaaaaaaa's picture
Upload 52 files
a797f9a verified
Raw History Blame Contribute Delete
9.47 kB
# scripts/pull_phasmophobia_wiki.py
from __future__ import annotations
import json
import re
import time
from pathlib import Path
from urllib.parse import quote
import requests
from bs4 import BeautifulSoup
from tqdm.auto import tqdm
API_URL = "https://phasmophobia.fandom.com/api.php"
BASE_URL = "https://phasmophobia.fandom.com/wiki"
OUT_DIR = Path("Resources/Documents")
OUT_DOCS = OUT_DIR / "phasmophobia_wiki_docs.jsonl"
OUT_CHUNKS = OUT_DIR / "phasmophobia_wiki_chunks.jsonl"
SESSION = requests.Session()
SESSION.headers.update({
"User-Agent": "MultimodalReasoningBot/0.1 clean-document-pull"
})
REMOVE_SELECTORS = [
"script",
"style",
"noscript",
"figure",
"img",
"audio",
"video",
"iframe",
"sup.reference",
".reference",
".references",
".reflist",
".mw-editsection",
".mw-empty-elt",
".toc",
"#toc",
".portable-infobox",
".infobox",
".navbox",
".metadata",
".ambox",
".catlinks",
".printfooter",
".noprint",
".thumb",
".gallery",
".wikia-gallery",
".pi-image",
".pi-data",
".pi-header",
".pi-title",
".page-header",
".page-footer",
".license-description",
".WikiaArticleFooter",
".rail-module",
]
SKIP_TITLE_PREFIXES = (
"User:",
"User talk:",
"Talk:",
"File:",
"Template:",
"Category:",
"Help:",
"Forum:",
"Blog:",
"MediaWiki:",
"Module:",
)
SKIP_EXACT_TITLES = {
"Main Page",
}
BAD_LINE_PATTERNS = [
r"^advertisement$",
r"^contents$",
r"^categories$",
r"^references$",
r"^gallery$",
r"^trivia$",
r"^see also$",
r"^external links$",
r"^community content is available",
r"^fandom apps",
r"^take your favorite fandoms",
r"^explore properties",
r"^view mobile site",
r"^follow on",
r"^sign in",
r"^create a free account",
]
def api_get(params: dict) -> dict:
params = {
"format": "json",
"formatversion": "2",
**params,
}
for attempt in range(5):
r = SESSION.get(API_URL, params=params, timeout=30)
if r.status_code == 429:
time.sleep(2 + attempt)
continue
r.raise_for_status()
return r.json()
raise RuntimeError(f"API failed after retries: {params}")
def get_all_pages() -> list[dict]:
pages = []
apcontinue = None
while True:
params = {
"action": "query",
"list": "allpages",
"apnamespace": 0,
"aplimit": "max",
"apfilterredir": "nonredirects",
}
if apcontinue:
params["apcontinue"] = apcontinue
data = api_get(params)
pages.extend(data.get("query", {}).get("allpages", []))
cont = data.get("continue", {})
apcontinue = cont.get("apcontinue")
if not apcontinue:
break
clean = []
for page in pages:
title = page["title"].strip()
if title in SKIP_EXACT_TITLES:
continue
if title.startswith(SKIP_TITLE_PREFIXES):
continue
clean.append(page)
return clean
def get_parsed_html(pageid: int) -> str | None:
data = api_get({
"action": "parse",
"pageid": pageid,
"prop": "text|displaytitle",
"redirects": "1",
"disableeditsection": "1",
"disabletoc": "1",
})
parsed = data.get("parse")
if not parsed:
return None
text = parsed.get("text")
if isinstance(text, dict):
return text.get("*")
return text
def clean_line(line: str) -> str:
line = re.sub(r"\[\s*edit\s*\]", "", line, flags=re.I)
line = re.sub(r"\[\d+\]", "", line)
line = re.sub(r"\s+", " ", line)
return line.strip()
def is_bad_line(line: str) -> bool:
if not line:
return True
low = line.lower().strip()
if len(low) <= 1:
return True
for pattern in BAD_LINE_PATTERNS:
if re.search(pattern, low):
return True
if low.startswith("http://") or low.startswith("https://"):
return True
if "retrieved from" in low:
return True
if "fandom.com" in low and len(low.split()) < 16:
return True
return False
def html_to_clean_text(html: str, keep_tables: bool = False) -> str:
soup = BeautifulSoup(html, "lxml")
root = soup.select_one(".mw-parser-output")
if root is None:
root = soup
for selector in REMOVE_SELECTORS:
for tag in root.select(selector):
tag.decompose()
if not keep_tables:
for tag in root.find_all("table"):
tag.decompose()
blocks = []
for tag in root.find_all(["h2", "h3", "h4", "p", "li"]):
text = clean_line(tag.get_text(" ", strip=True))
if is_bad_line(text):
continue
if tag.name in {"h2", "h3", "h4"}:
# Keep useful section boundaries, but not table-of-contents garbage.
blocks.append(f"\n## {text}\n")
else:
blocks.append(text)
text = "\n".join(blocks)
text = re.sub(r"\n{3,}", "\n\n", text)
text = re.sub(r"[ \t]{2,}", " ", text)
return text.strip()
def title_to_url(title: str) -> str:
return f"{BASE_URL}/{quote(title.replace(' ', '_'))}"
def pull_docs(limit: int | None = None, test_pages: list[str] | None = None) -> list[dict]:
OUT_DIR.mkdir(parents=True, exist_ok=True)
if test_pages:
all_pages = get_all_pages()
wanted = {x.lower() for x in test_pages}
pages = [p for p in all_pages if p["title"].lower() in wanted]
else:
pages = get_all_pages()
if limit is not None:
pages = pages[:limit]
print(f"[Pull] Pages selected: {len(pages)}")
records = []
for page in tqdm(pages, desc="Pulling Phasmophobia wiki"):
pageid = int(page["pageid"])
title = page["title"].strip()
try:
html = get_parsed_html(pageid)
if not html:
continue
text = html_to_clean_text(html, keep_tables=False)
# Skip pages that are only nav/category/etc.
if len(text.split()) < 40:
continue
records.append({
"source": "phasmophobia.fandom.com",
"pageid": pageid,
"title": title,
"url": title_to_url(title),
"text": text,
})
time.sleep(0.05)
except Exception as e:
print(f"[WARN] Failed {title}: {e}")
return records
def chunk_text(text: str, chunk_words: int = 256, overlap: int = 48) -> list[str]:
words = text.split()
if len(words) <= chunk_words:
return [" ".join(words)]
chunks = []
step = max(1, chunk_words - overlap)
for start in range(0, len(words), step):
piece = words[start:start + chunk_words]
if len(piece) < 40:
continue
chunks.append(" ".join(piece))
return chunks
def save_docs(records: list[dict]) -> None:
OUT_DIR.mkdir(parents=True, exist_ok=True)
with OUT_DOCS.open("w", encoding="utf-8") as f:
for record in records:
f.write(json.dumps(record, ensure_ascii=False) + "\n")
print(f"[Save] Docs: {OUT_DOCS} ({len(records)} records)")
def save_chunks(records: list[dict], chunk_words: int = 256, overlap: int = 48) -> None:
count = 0
with OUT_CHUNKS.open("w", encoding="utf-8") as f:
for record in records:
chunks = chunk_text(
record["text"],
chunk_words=chunk_words,
overlap=overlap,
)
for chunk_id, chunk in enumerate(chunks):
f.write(json.dumps({
"source": record["source"],
"title": record["title"],
"url": record["url"],
"pageid": record["pageid"],
"chunk_id": chunk_id,
"text": chunk,
}, ensure_ascii=False) + "\n")
count += 1
print(f"[Save] Chunks: {OUT_CHUNKS} ({count} chunks)")
def preview(records: list[dict], max_docs: int = 5, chars: int = 1200) -> None:
for record in records[:max_docs]:
print("=" * 100)
print(record["title"])
print(record["url"])
print("-" * 100)
print(record["text"][:chars])
def quality_check(records: list[dict]) -> None:
bad_markers = [
"Advertisement",
"Fandom Apps",
"Take your favorite fandoms",
"View Mobile Site",
"Create a Free Account",
"Sign In",
"Retrieved from",
]
print("\n[Quality Check]")
for marker in bad_markers:
hits = [
r["title"]
for r in records
if marker.lower() in r["text"].lower()
]
print(f"{marker!r}: {len(hits)} hits")
if hits[:5]:
print(" sample:", hits[:5])
def main():
# Test first against known pages I checked manually:
# Phasmophobia, Ghost, Map
test = False
if test:
records = pull_docs(test_pages=["Phasmophobia", "Ghost", "Map"])
else:
records = pull_docs()
save_docs(records)
save_chunks(records, chunk_words=256, overlap=48)
quality_check(records)
preview(records, max_docs=3)
if __name__ == "__main__":
main()