Spaces:
Running
Running
| """Turn every collector's output into `Doc`s. | |
| One adapter per source file. The adapters are the only place in the runtime that | |
| knows a collector's field names — everything downstream sees `Doc`. | |
| Three jobs beyond renaming fields: | |
| 1. **Cross-source event dedupe.** AnchorLink and the LiveWhale campus feed are the | |
| same platform underneath and overlap heavily (82 of AnchorLink's 92 events are | |
| also in the campus feed). `collectors/events/README.md` states plainly that the | |
| runtime has to dedupe at read time, and this is that. | |
| 2. **Occurrence collapse.** The campus feed publishes one row per *occurrence* — | |
| 633 rows for 534 events, one exhibition repeating 26 times. That's correct data | |
| and the wrong search result, so one `Doc` per event carries its occurrence dates | |
| in `extra["occurrences"]`. | |
| 3. **Resource attachment.** `resources.json` files hold the "how to engage" layer, | |
| but `study-abroad/resources.json` is 134 near-identical budget-sheet links, one | |
| per program. Those attach to their program; only genuinely distinct resources | |
| (portals, advising, forms) become documents of their own. | |
| Deduplication that already happened *at build time* is trusted, not redone: | |
| duplicate directory records, duplicate map pins, and academic-calendar term | |
| duplication were all settled by the collectors. Their merge artifacts (`alt_ids`, | |
| `aliases`) are folded into the searchable text so "Jacobs Hall" still reaches | |
| Featheringill. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import logging | |
| import re | |
| from dataclasses import replace | |
| from pathlib import Path | |
| from . import topics | |
| from .doc import Doc | |
| log = logging.getLogger("foresight.kb") | |
| # Files that exist but must never be indexed. | |
| # *_needs_review.json — a human queue of unresolved records, not student-facing facts | |
| # structure.json — the catalog PDF's section tree; wayfinding for the collector | |
| # collection_meta — provenance | |
| # collection_meta.json — provenance, not content | |
| # structure.json — the catalog PDF's section tree; collector wayfinding | |
| # needs_review.json — NOT a list of records to exclude. It annotates the 447 | |
| # directory entities whose *location* couldn't be confidently resolved; every | |
| # one of them is also in directory.json with `building_slug: null` and a real | |
| # phone, email and website. Dropping them would cost the companion 447 real | |
| # offices. They're indexed, and flagged `location_unknown` so an answer can | |
| # give the contact without inventing a room. | |
| NEVER_INDEXED = ("collection_meta.json", "structure.json", "needs_review.json") | |
| def _load(root: Path, rel: str): | |
| f = root / rel | |
| if not f.exists(): | |
| log.warning("kb: %s is missing — skipping", rel) | |
| return [] | |
| try: | |
| return json.loads(f.read_text(encoding="utf-8")) | |
| except (json.JSONDecodeError, OSError) as err: | |
| log.error("kb: %s failed to parse (%s) — skipping", rel, err) | |
| return [] | |
| def _join(*parts) -> str: | |
| """Searchable blob from a mixed bag of strings, lists and Nones.""" | |
| out: list[str] = [] | |
| for p in parts: | |
| if not p: | |
| continue | |
| if isinstance(p, (list, tuple)): | |
| out.extend(str(x) for x in p if x) | |
| elif isinstance(p, dict): | |
| out.extend(f"{k} {v}" for k, v in p.items() if v) | |
| else: | |
| out.append(str(p)) | |
| return " ".join(out) | |
| def _domains(rec) -> tuple[str, ...]: | |
| return tuple(rec.get("domains") or ()) | |
| # `extra["audience"]` means three different things depending on `kind`, because three | |
| # collectors independently chose the obvious word. Anything reading it must scope by | |
| # kind — `topics.is_for_students` is only ever handed an event's: | |
| # | |
| # event list[str] from `topics.AUDIENCES` — who may/should attend | |
| # ("undergraduate", "graduate", "faculty-staff", "alumni", "public") | |
| # office list[str] of who an office serves ("faculty", "staff", "student", | |
| # "admissions", "academics", "parents") | |
| # process a bare string naming a cohort ("first_year", "all", "returning") | |
| # | |
| # Only the event one is a controlled vocabulary with a runtime filter attached. | |
| def _norm_title(t: str | None) -> str: | |
| return re.sub(r"[^a-z0-9]+", "", (t or "").lower()) | |
| # --- AnchorLink ------------------------------------------------------------- | |
| def anchorlink_orgs(root: Path) -> list[Doc]: | |
| docs = [] | |
| for r in _load(root, "anchorlink/organizations.json"): | |
| if r.get("status") and r["status"] != "Active": | |
| continue | |
| docs.append(Doc( | |
| id=f"anchorlink:org:{r['id']}", | |
| source="anchorlink", kind="organization", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("short_name"), r.get("summary"), | |
| r.get("description"), r.get("categories"), | |
| topics.labels(r.get("topics"))), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"categories": r.get("categories") or [], | |
| "topics": r.get("topics") or [], | |
| "audience": r.get("audience") or [], | |
| "short_name": r.get("short_name")}, | |
| )) | |
| return docs | |
| # --- Events (two sources, one index) ---------------------------------------- | |
| def events(root: Path) -> list[Doc]: | |
| """AnchorLink events + the LiveWhale campus feed, collapsed and deduped.""" | |
| docs: list[Doc] = [] | |
| seen_urls: set[str] = set() | |
| seen_titles: set[tuple[str, str]] = set() | |
| # AnchorLink first — it carries host and org linkage, so it wins a collision. | |
| for r in _load(root, "anchorlink/events.json"): | |
| if r.get("is_canceled"): | |
| continue | |
| url, day = r.get("url"), r.get("start_date") or "" | |
| docs.append(Doc( | |
| id=f"anchorlink:event:{r['id']}", | |
| source="anchorlink", kind="event", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("description"), r.get("categories"), | |
| r.get("hosts"), r.get("location"), | |
| topics.labels(r.get("topics"))), | |
| url=url, domains=_domains(r), | |
| start=r.get("start"), end=r.get("end"), | |
| extra={"location": r.get("location"), "hosts": r.get("hosts") or [], | |
| "topics": r.get("topics") or [], | |
| "audience": r.get("audience") or [], | |
| "contact_email": r.get("author_email"), | |
| "occurrences": [r.get("start")] if r.get("start") else []}, | |
| )) | |
| if url: | |
| seen_urls.add(url) | |
| seen_titles.add((_norm_title(r.get("title")), day)) | |
| # LiveWhale: group occurrences by event id before deduping against AnchorLink. | |
| by_event: dict = {} | |
| for r in _load(root, "events/events.json"): | |
| if r.get("is_canceled"): | |
| continue | |
| by_event.setdefault(r["id"], []).append(r) | |
| dropped = 0 | |
| for event_id, rows in by_event.items(): | |
| rows.sort(key=lambda r: r.get("date_iso") or "") | |
| first = rows[0] | |
| url = first.get("url") | |
| day = (first.get("date_iso") or "")[:10] | |
| if (url and url in seen_urls) or (_norm_title(first.get("title")), day) in seen_titles: | |
| dropped += 1 | |
| continue | |
| docs.append(Doc( | |
| id=f"events:event:{event_id}", | |
| source="events", kind="event", | |
| title=first.get("title") or "", | |
| text=_join(first.get("title"), first.get("description"), first.get("tags"), | |
| first.get("group_title"), first.get("location"), | |
| topics.labels(first.get("topics"))), | |
| url=url, domains=_domains(first), | |
| start=first.get("date_iso"), end=rows[-1].get("end_iso") or rows[-1].get("date_iso"), | |
| extra={"location": first.get("location"), | |
| "topics": first.get("topics") or [], | |
| "audience": first.get("audience") or [], | |
| # The contact published *for the event*, not the listing owner — | |
| # they disagree on 437 of 447 records and this is the one to show. | |
| "contact_email": first.get("contact_email"), | |
| "contact_info": first.get("contact_info"), | |
| "cost": first.get("cost"), | |
| "has_registration": first.get("has_registration"), | |
| "occurrences": [r.get("date_iso") for r in rows if r.get("date_iso")]}, | |
| )) | |
| log.info("kb: events — %d docs (%d LiveWhale events deduped against AnchorLink)", | |
| len(docs), dropped) | |
| return docs | |
| # --- Academics -------------------------------------------------------------- | |
| def courses(root: Path) -> list[Doc]: | |
| """Catalog course details, topped up with Kuali-only courses.""" | |
| docs, seen = [], set() | |
| for r in _load(root, "academics/catalog/course_details.json"): | |
| cid = r.get("id") | |
| if not cid: | |
| continue | |
| seen.add(cid) | |
| tags = r.get("tags") or {} | |
| docs.append(Doc( | |
| id=f"academics:course:{cid}", | |
| source="academics", kind="course", | |
| title=f"{cid} — {r.get('title') or ''}".strip(" —"), | |
| text=_join(cid, r.get("title"), r.get("subject_name"), r.get("description"), | |
| tags.get("axle"), tags.get("core"), tags.get("le")), | |
| url=(r.get("source") or {}).get("url"), domains=_domains(r), | |
| extra={"course_id": cid, "subject": r.get("subject"), | |
| "subject_name": r.get("subject_name"), | |
| "credit_hours": r.get("credit_hours"), | |
| "axle": tags.get("axle") or [], "core": tags.get("core") or [], | |
| "relations": r.get("relations") or {}, | |
| "description": r.get("description")}, | |
| )) | |
| # Courses in Kuali but not in the printed catalog — identity only, no description. | |
| for r in _load(root, "academics/courses.json"): | |
| cid = r.get("course_id") | |
| if not cid or cid in seen: | |
| continue | |
| seen.add(cid) | |
| docs.append(Doc( | |
| id=f"academics:course:{cid}", | |
| source="academics", kind="course", | |
| title=f"{cid} — {r.get('title') or ''}".strip(" —"), | |
| text=_join(cid, r.get("title"), r.get("subject_name")), | |
| url=None, domains=_domains(r), | |
| extra={"course_id": cid, "subject": r.get("subject"), | |
| "subject_name": r.get("subject_name")}, | |
| )) | |
| return docs | |
| def programs(root: Path) -> list[Doc]: | |
| """Kuali program list + the catalog's requirement trees, merged by title.""" | |
| docs = [] | |
| reqs_by_title = {} | |
| for r in _load(root, "academics/catalog/program_requirements.json"): | |
| reqs_by_title[_norm_title(r.get("title"))] = r | |
| used = set() | |
| for r in _load(root, "academics/programs.json"): | |
| key = _norm_title(r.get("title")) | |
| req = reqs_by_title.get(key) | |
| if req: | |
| used.add(key) | |
| rule_text = _join(*[g.get("rule_text") for g in (req or {}).get("requirement_groups", [])]) | |
| docs.append(Doc( | |
| id=f"academics:program:{r.get('code') or key}", | |
| source="academics", kind="program", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("kind"), r.get("schools"), rule_text), | |
| url=r.get("api_url"), domains=_domains(r), | |
| extra={"program_kind": r.get("kind"), "schools": r.get("schools") or [], | |
| "requirements": (req or {}).get("requirement_groups"), | |
| "total_hours": (req or {}).get("total_hours")}, | |
| )) | |
| # Requirement trees with no Kuali counterpart (combined/joint degrees, mostly). | |
| for key, req in reqs_by_title.items(): | |
| if key in used: | |
| continue | |
| rule_text = _join(*[g.get("rule_text") for g in req.get("requirement_groups", [])]) | |
| docs.append(Doc( | |
| id=f"academics:program:{req.get('id')}", | |
| source="academics", kind="program", | |
| title=req.get("title") or "", | |
| text=_join(req.get("title"), req.get("kind"), req.get("school"), rule_text), | |
| url=None, domains=("vu",), | |
| extra={"program_kind": req.get("kind"), | |
| "schools": [req.get("school")] if req.get("school") else [], | |
| "requirements": req.get("requirement_groups"), | |
| "total_hours": req.get("total_hours")}, | |
| )) | |
| return docs | |
| def academic_misc(root: Path) -> list[Doc]: | |
| docs = [] | |
| for r in _load(root, "academics/catalog/policies.json"): | |
| docs.append(Doc( | |
| id=f"academics:policy:{r.get('id')}", | |
| source="academics", kind="policy", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("summary"), r.get("content")), | |
| url=None, domains=_domains(r) or ("vu",), | |
| extra={"policy_type": r.get("type"), "rules": r.get("rules"), | |
| "summary": r.get("summary"), "entry_year": r.get("entry_year")}, | |
| )) | |
| for r in _load(root, "academics/calendar.json"): | |
| docs.append(Doc( | |
| id=f"academics:calendar:{r.get('occurrence_id') or r.get('id')}", | |
| source="academics", kind="key_date", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("description"), r.get("terms"), r.get("tags")), | |
| url=r.get("url"), domains=_domains(r) or ("vu",), | |
| start=r.get("date_iso"), end=r.get("end_iso"), | |
| extra={"terms": r.get("terms") or [], "date": r.get("date")}, | |
| )) | |
| for r in _load(root, "academics/catalog/liberal_ed_index.json"): | |
| docs.append(Doc( | |
| id=f"academics:liberal_ed:{r.get('curriculum')}:{r.get('category')}", | |
| source="academics", kind="liberal_ed", | |
| title=f"{r.get('curriculum')} — {r.get('category_name')}", | |
| # The 685 course ids stay out of the searchable text and live in extra; | |
| # indexing them would let one category match almost any course query. | |
| text=_join(r.get("curriculum"), r.get("curriculum_name"), | |
| r.get("category"), r.get("category_name")), | |
| url=None, domains=("vu",), | |
| extra={"curriculum": r.get("curriculum"), "category": r.get("category"), | |
| "count": r.get("count"), "courses": r.get("courses") or []}, | |
| )) | |
| for r in _load(root, "academics/campuses.json"): | |
| docs.append(Doc( | |
| id=f"academics:campus:{r.get('slug')}", | |
| source="academics", kind="page", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("summary"), r.get("content")), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"campus": r.get("campus")}, | |
| )) | |
| return docs | |
| # --- YES class schedule (per-term offerings) --------------------------------- | |
| def yes_offerings(root: Path) -> list[Doc]: | |
| """What is actually offered each term, from the public YES class search. | |
| One Doc per course per term — sections stay inline in `extra` rather than | |
| becoming documents, or MATH 1200's twenty-five sections would fill every | |
| result list. The term name is in the *title* deliberately: "fall" in a query | |
| then title-boosts the right term's offering. | |
| Distinct from `academics` courses: that source is the catalog of everything | |
| that *exists*; this one is what actually *runs* in a term, with meeting | |
| times, instructors, and an enrollment snapshot (`fetched_at` says how old). | |
| """ | |
| yes_root = root / "yes" | |
| if not yes_root.is_dir(): | |
| return [] | |
| docs = [] | |
| for f in sorted(yes_root.glob("*/*.json")): | |
| for r in _load(root, f.relative_to(root).as_posix()): | |
| sections = r.get("sections") or [] | |
| instructors = sorted({i for s in sections | |
| for i in (s.get("instructors") or [])}) | |
| docs.append(Doc( | |
| id=f"yes:offering:{r.get('term_code')}:" | |
| f"{(r.get('course_id') or '').replace(' ', '')}", | |
| source="yes", kind="class_offering", | |
| title=f"{r.get('course_id')} — {r.get('title') or ''} " | |
| f"({r.get('term_name')})", | |
| text=_join(r.get("course_id"), r.get("title"), | |
| r.get("subject_name"), r.get("term_name"), | |
| r.get("career"), r.get("school"), r.get("description"), | |
| r.get("attributes"), instructors), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"course_id": r.get("course_id"), | |
| "subject": r.get("subject"), | |
| "subject_name": r.get("subject_name"), | |
| "term_code": r.get("term_code"), | |
| "term_name": r.get("term_name"), | |
| "career": r.get("career"), "school": r.get("school"), | |
| "credit_hours": r.get("credit_hours"), | |
| "session": r.get("session"), | |
| "session_start": r.get("session_start"), | |
| "session_end": r.get("session_end"), | |
| "description": r.get("description"), | |
| "requirements": r.get("requirements"), | |
| "attributes": r.get("attributes") or [], | |
| "notes": r.get("notes") or [], | |
| "sections": sections, | |
| "fetched_at": r.get("fetched_at")}, | |
| )) | |
| return docs | |
| # --- Places: directory + buildings ------------------------------------------ | |
| def places(root: Path) -> list[Doc]: | |
| """Offices and buildings. Build-time dedupe is trusted; `alt_ids` and | |
| `aliases` are folded into the text so merged names stay findable.""" | |
| docs = [] | |
| for r in _load(root, "directory/directory.json"): | |
| docs.append(Doc( | |
| id=f"directory:office:{r.get('id')}", | |
| source="directory", kind="office", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("alt_ids"), r.get("kind"), | |
| r.get("building_name"), r.get("audience"), r.get("evidence_quote")), | |
| url=r.get("website"), domains=_domains(r), | |
| extra={"building_slug": r.get("building_slug"), | |
| "building_name": r.get("building_name"), "room": r.get("room"), | |
| "address": r.get("address"), "phone": r.get("phone"), | |
| "email": r.get("email"), "office_kind": r.get("kind"), | |
| "audience": r.get("audience") or [], "verified": r.get("verified"), | |
| # The collector refuses to guess a building rather than get one | |
| # wrong. Carry that forward so an answer says "I have their | |
| # email but not a room" instead of inventing one. | |
| "location_unknown": not r.get("building_slug")}, | |
| )) | |
| for r in _load(root, "buildings/pages.json"): | |
| docs.append(Doc( | |
| id=f"buildings:building:{r.get('slug')}", | |
| source="buildings", kind="building", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("aliases"), r.get("address"), r.get("description")), | |
| url=r.get("website") or r.get("map_url"), domains=(), | |
| extra={"slug": r.get("slug"), "address": r.get("address"), | |
| "lat": r.get("lat"), "lng": r.get("lng"), | |
| "map_url": r.get("map_url"), "aliases": r.get("aliases") or []}, | |
| )) | |
| return docs | |
| # --- Residential ------------------------------------------------------------ | |
| def residential(root: Path) -> list[Doc]: | |
| docs = [] | |
| for r in _load(root, "residential/houses.json"): | |
| docs.append(Doc( | |
| id=f"residential:house:{_norm_title(r.get('name'))}", | |
| source="residential", kind="house", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("type"), r.get("description"), | |
| r.get("namesake"), r.get("neighborhood"), r.get("class_years")), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"house_type": r.get("type"), "class_years": r.get("class_years") or [], | |
| "room_types": r.get("room_types") or [], "location": r.get("location"), | |
| "building_slug": r.get("building_slug")}, | |
| )) | |
| for r in _load(root, "residential/contacts.json"): | |
| c = r.get("contact") or {} | |
| docs.append(Doc( | |
| id=f"residential:contact:{r.get('role')}", | |
| source="residential", kind="contact", | |
| title=(r.get("scenario") or r.get("role") or "").capitalize(), | |
| text=_join(r.get("role"), r.get("scenario"), c.get("office"), c.get("portal")), | |
| url=c.get("url"), domains=_domains(r), | |
| extra={"role": r.get("role"), "scenario": r.get("scenario"), | |
| "phone": c.get("phone"), "email": c.get("email"), | |
| "hours": c.get("hours"), "office": c.get("office")}, | |
| )) | |
| for r in _load(root, "residential/processes.json"): | |
| docs.append(Doc( | |
| id=f"residential:process:{r.get('process')}", | |
| source="residential", kind="process", | |
| title=(r.get("process") or "").replace("_", " ").capitalize(), | |
| text=_join(r.get("process"), r.get("audience"), r.get("summary"), | |
| *[s.get("description") for s in (r.get("steps") or [])]), | |
| url=r.get("source_url"), domains=_domains(r), | |
| extra={"audience": r.get("audience"), "summary": r.get("summary"), | |
| "steps": r.get("steps") or [], "key_dates": r.get("key_dates")}, | |
| )) | |
| for r in _load(root, "residential/policies.json"): | |
| docs.append(Doc( | |
| id=f"residential:policy:{_norm_title(r.get('title'))}", | |
| source="residential", kind="policy", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("category"), r.get("summary"), r.get("detail")), | |
| url=r.get("source_url"), domains=_domains(r), | |
| extra={"category": r.get("category"), "summary": r.get("summary")}, | |
| )) | |
| return docs | |
| # --- Recreation & sports ---------------------------------------------------- | |
| def recreation(root: Path) -> list[Doc]: | |
| """The Rec Center, intramurals, club sports, and varsity games. | |
| `how_to_participate` is deliberately first in the searchable text: the whole | |
| point of this source is that a student asking "how do I sign up for | |
| intramural basketball" gets procedure, not a phone number (issue #46). | |
| """ | |
| docs = [] | |
| for r in _load(root, "recreation/programs.json"): | |
| docs.append(Doc( | |
| id=f"recreation:program:{r.get('id')}", | |
| source="recreation", kind=r.get("type") or "program", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("how_to_participate"), r.get("summary"), | |
| # the words students actually use ("gym", "free", "sign up"), | |
| # which the source pages frequently don't contain | |
| r.get("search_terms"), | |
| r.get("eligibility"), r.get("exclusions"), r.get("cost_note"), | |
| r.get("hours"), r.get("hours_text"), r.get("administered_by"), | |
| *(r.get("sports") or []), | |
| *(f"{lt.get('name')} {lt.get('description')}" | |
| for lt in (r.get("league_types") or []))), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"category": r.get("category"), | |
| "how_to_participate": r.get("how_to_participate"), | |
| "requires_login": r.get("requires_login"), | |
| "registration_url": r.get("registration_url"), | |
| "cost": r.get("cost"), "eligibility": r.get("eligibility"), | |
| "building_slug": r.get("building_slug"), | |
| "contact": r.get("contact"), "hours_url": r.get("hours_url")}, | |
| )) | |
| for r in _load(root, "recreation/club_sports.json"): | |
| docs.append(Doc( | |
| id=f"recreation:club:{r.get('id')}", | |
| source="recreation", kind="club_sport", | |
| title=f"{r.get('name')} (club sport)", | |
| text=_join(r.get("name"), "club sport team", r.get("summary"), | |
| r.get("how_to_participate"), r.get("eligibility")), | |
| url=r.get("anchorlink_url") or r.get("url"), domains=_domains(r), | |
| extra={"category": r.get("category"), | |
| "anchorlink_org_id": r.get("anchorlink_org_id"), | |
| "how_to_participate": r.get("how_to_participate"), | |
| "building_slug": r.get("building_slug"), | |
| "contact": r.get("contact")}, | |
| )) | |
| for r in _load(root, "recreation/games.json"): | |
| home = "home" if r.get("is_home") else "away" | |
| free = "free student tickets" if r.get("tickets_free_to_students") else "" | |
| docs.append(Doc( | |
| id=f"recreation:game:{r.get('id')}", | |
| source="recreation", kind="game", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("sport"), home, "game", | |
| r.get("location"), r.get("time_note"), free, r.get("ticket_note")), | |
| url=r.get("url"), domains=_domains(r), | |
| start=r.get("date_iso"), end=r.get("date_iso"), | |
| extra={"category": r.get("category"), "sport": r.get("sport"), | |
| "opponent": r.get("opponent"), "is_home": r.get("is_home"), | |
| "start_time": r.get("start_time"), "time_note": r.get("time_note"), | |
| "location": r.get("location"), "building_slug": r.get("building_slug"), | |
| "tickets_free_to_students": r.get("tickets_free_to_students"), | |
| "ticket_url": r.get("ticket_url"), "ticket_note": r.get("ticket_note")}, | |
| )) | |
| for r in _load(root, "recreation/calendar.json"): | |
| docs.append(Doc( | |
| id=f"recreation:im:{r.get('id')}", | |
| source="recreation", kind="registration_window", | |
| title=f"{r.get('sport')} — intramural registration", | |
| text=_join(r.get("sport"), "intramural registration deadline league", | |
| r.get("season"), r.get("how_to_participate")), | |
| url=r.get("url"), domains=_domains(r), | |
| start=r.get("registration_opens") or r.get("date_iso"), | |
| end=r.get("end_iso"), | |
| extra={"category": r.get("category"), "sport": r.get("sport"), | |
| "registration_opens": r.get("registration_opens"), | |
| "registration_closes": r.get("registration_closes"), | |
| "season": r.get("season"), | |
| # a deadline without the procedure is a date the student | |
| # can't act on — carry the how-to onto the dated record too | |
| "how_to_participate": r.get("how_to_participate"), | |
| "requires_login": r.get("requires_login"), | |
| "registration_url": r.get("registration_url")}, | |
| )) | |
| for r in _load(root, "recreation/fusion_products.json"): | |
| docs.append(Doc( | |
| id=f"recreation:class:{_norm_title(r.get('name'))}", | |
| source="recreation", kind="class", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("classification"), r.get("day"), | |
| r.get("time"), r.get("location")), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"classification": r.get("classification"), "day": r.get("day"), | |
| "time": r.get("time"), "location": r.get("location"), | |
| "cost": None, "requires_login": r.get("requires_login"), | |
| "registration_url": r.get("registration_url")}, | |
| )) | |
| return docs | |
| # --- Funding ---------------------------------------------------------------- | |
| def funding(root: Path) -> list[Doc]: | |
| docs = [] | |
| for r in _load(root, "funding/awards.json"): | |
| # An award is listed once per applicant level — Lafayette appears for | |
| # postgraduates and for graduate students, with different deadlines. Those | |
| # are distinct records, so the level belongs in the id. | |
| level = r.get("applicant_level") or "any" | |
| docs.append(Doc( | |
| id=f"funding:award:{_norm_title(r.get('name'))}:{level}", | |
| source="funding", kind="award", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("description"), r.get("eligibility"), | |
| r.get("level_label"), r.get("tier")), | |
| url=r.get("official_url") or r.get("source_url"), domains=_domains(r), | |
| extra={"eligibility": r.get("eligibility"), "tier": r.get("tier"), | |
| "campus_deadline": r.get("campus_deadline"), | |
| "priority_deadline": r.get("priority_deadline"), | |
| "applicant_level": level, "level_label": r.get("level_label")}, | |
| )) | |
| for r in _load(root, "funding/scholarships.json"): | |
| docs.append(Doc( | |
| id=f"funding:scholarship:{_norm_title(r.get('name'))}", | |
| source="funding", kind="award", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("description"), r.get("award"), r.get("type")), | |
| url=r.get("url") or r.get("source_url"), domains=_domains(r), | |
| extra={"award": r.get("award"), "campus_deadline": r.get("deadline"), | |
| "scholarship_type": r.get("type")}, | |
| )) | |
| for r in _load(root, "funding/research_programs.json"): | |
| docs.append(Doc( | |
| id=f"funding:research:{_norm_title(r.get('name'))}", | |
| source="funding", kind="research_program", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("description"), r.get("eligibility_group")), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"stipend": r.get("stipend"), "deadline": r.get("deadline"), | |
| "eligibility_group": r.get("eligibility_group")}, | |
| )) | |
| return docs | |
| # --- Prose pages (immersion, study abroad, funding, residential) ------------ | |
| PAGE_FILES = [ | |
| ("immersion/pages.json", "immersion"), | |
| ("study-abroad/pages.json", "study-abroad"), | |
| ("study-abroad/summer_pages.json", "study-abroad"), | |
| ("funding/pages.json", "funding"), | |
| ("residential/pages.json", "residential"), | |
| ("recreation/pages.json", "recreation"), | |
| ] | |
| def pages(root: Path) -> list[Doc]: | |
| docs = [] | |
| for rel, source in PAGE_FILES: | |
| for r in _load(root, rel): | |
| docs.append(Doc( | |
| id=f"{source}:page:{r.get('slug') or _norm_title(r.get('title'))}", | |
| source=source, kind="page", | |
| title=r.get("title") or "", | |
| text=_join(r.get("title"), r.get("summary"), r.get("content")), | |
| url=r.get("url"), domains=_domains(r), | |
| extra={"page_kind": r.get("kind"), "section": r.get("section"), | |
| "summary": r.get("summary")}, | |
| )) | |
| return docs | |
| def study_abroad_programs(root: Path) -> list[Doc]: | |
| docs = [] | |
| for r in _load(root, "study-abroad/programs.json"): | |
| stats = r.get("stats") or {} | |
| docs.append(Doc( | |
| id=f"study-abroad:program:{r.get('program_id')}", | |
| source="study-abroad", kind="program", | |
| title=r.get("name") or "", | |
| text=_join(r.get("name"), r.get("city"), r.get("country"), | |
| r.get("overview"), stats), | |
| url=r.get("url"), domains=_domains(r) or ("crew", "future"), | |
| extra={"program_id": r.get("program_id"), "city": r.get("city"), | |
| "country": r.get("country"), "stats": stats}, | |
| )) | |
| return docs | |
| # --- The "how to engage" layer ---------------------------------------------- | |
| RESOURCE_FILES = [ | |
| ("immersion/resources.json", "immersion"), | |
| ("study-abroad/resources.json", "study-abroad"), | |
| ("funding/resources.json", "funding"), | |
| ("residential/resources.json", "residential"), | |
| ("academics/resources.json", "academics"), | |
| ("recreation/resources.json", "recreation"), | |
| ] | |
| def resources(root: Path, by_id: dict[str, Doc]) -> list[Doc]: | |
| """Application portals, advising links, forms, class search. | |
| A resource whose every `found_on` is a specific program is an attachment to | |
| that program, not a document — `study-abroad/resources.json` is 134 copies of | |
| "See Budget Sheets", one per program, and indexing them would swamp search. | |
| """ | |
| docs, seen_urls = [], set() | |
| attach: dict[str, list[dict]] = {} | |
| for rel, source in RESOURCE_FILES: | |
| for r in _load(root, rel): | |
| url, found_on = r.get("url"), r.get("found_on") or [] | |
| entry = {"text": r.get("text"), "url": url, "kind": r.get("kind")} | |
| program_refs = [f for f in found_on if str(f).startswith("program:")] | |
| if program_refs and len(program_refs) == len(found_on): | |
| for ref in program_refs: | |
| attach.setdefault(f"{source}:program:{ref.split(':', 1)[1]}", []).append(entry) | |
| continue | |
| if not url or url in seen_urls: | |
| continue | |
| seen_urls.add(url) | |
| docs.append(Doc( | |
| id=f"{source}:resource:{len(seen_urls)}", | |
| source=source, kind="resource", | |
| title=r.get("text") or url, | |
| text=_join(r.get("text"), r.get("kind"), found_on, source), | |
| url=url, domains=(), | |
| extra={"resource_kind": r.get("kind"), "found_on": found_on}, | |
| )) | |
| # Fold program-specific resources onto their program (extras are mutable by design). | |
| for doc_id, entries in attach.items(): | |
| target = by_id.get(doc_id) | |
| if target is not None: | |
| target.extra.setdefault("resources", []).extend(entries) | |
| return docs | |
| # --- Entry point ------------------------------------------------------------ | |
| def load_all(root: Path) -> list[Doc]: | |
| docs: list[Doc] = [] | |
| for fn in (anchorlink_orgs, events, courses, programs, academic_misc, | |
| yes_offerings, places, residential, recreation, funding, pages, | |
| study_abroad_programs): | |
| try: | |
| docs.extend(fn(root)) | |
| except Exception: # one bad source shouldn't kill the app | |
| log.exception("kb: adapter %s failed — continuing without it", fn.__name__) | |
| by_id = {d.id: d for d in docs} | |
| try: | |
| docs.extend(resources(root, by_id)) | |
| except Exception: | |
| log.exception("kb: resource adapter failed — continuing without it") | |
| # Ids must be unique — they're what a tool result cites back — but a collision | |
| # means the id scheme is too coarse, not that the record is redundant. Suffix | |
| # rather than drop, and say so, so a real duplicate shows up in the log instead | |
| # of silently costing the student an answer. | |
| seen: set[str] = set() | |
| out: list[Doc] = [] | |
| collisions = 0 | |
| for d in docs: | |
| if d.id in seen: | |
| collisions += 1 | |
| n = 2 | |
| while f"{d.id}#{n}" in seen: | |
| n += 1 | |
| d = replace(d, id=f"{d.id}#{n}") | |
| seen.add(d.id) | |
| out.append(d) | |
| if collisions: | |
| log.warning("kb: %d documents needed an id suffix — the id scheme is too " | |
| "coarse for those sources", collisions) | |
| return out | |