cascade_risk / src /eval /news_loader.py
Lucasoppem's picture
Sync from GitHub main (part 2)
36f9d47 verified
Raw
History Blame Contribute Delete
1.42 kB
"""Load scraped news articles for a given event_id."""
from __future__ import annotations
import json
import logging
from pathlib import Path
logger = logging.getLogger(__name__)
def load_articles(
event_id: str,
articles_dir: Path,
max_articles: int,
max_chars_per_article: int,
) -> list[dict]:
"""Return article dicts for this event, sorted by date ascending, truncated.
Each returned dict has keys: url, title, date, text. Articles missing the
`date` field are placed at the end of the list. The `text` key matches the
scraper's on-disk format (src/data/news_scraper.py writes "text").
"""
articles_dir = Path(articles_dir)
if not articles_dir.exists():
return []
matches = sorted(articles_dir.glob(f"{event_id}_*.json"))
articles: list[dict] = []
for path in matches:
try:
data = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as e:
logger.warning("Skipping unreadable article %s: %s", path, e)
continue
text = (data.get("text") or "")[:max_chars_per_article]
articles.append({
"url": data.get("url", ""),
"title": data.get("title", ""),
"date": data.get("date", ""),
"text": text,
})
articles.sort(key=lambda a: a.get("date") or "9999-99-99")
return articles[:max_articles]