Spaces:
Sleeping
Sleeping
File size: 5,909 Bytes
19de729 bfcc872 19de729 bfcc872 19de729 bfcc872 19de729 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 | # src/analyzer/data_loader_supporting.py
from __future__ import annotations
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, List, Optional, Iterable, Union
import json, re
from .utils.text import to_number
@dataclass
class SupportingDoc:
grant_id: str
url: str
title: str
open_date: Optional[str]
close_date: Optional[str]
notify_date: Optional[str]
funding_min: Optional[float]
funding_max: Optional[float]
total_pot: Optional[float]
funding_rates: Optional[str]
duration_min: Optional[int]
duration_max: Optional[int]
text: str # flattened blob for retrieval
sections: Dict[str, str] # raw sections, if present
# Number parsing moved to utils.text.to_number()
# Keeping wrapper for backward compatibility
def _num(x):
return to_number(x)
def _int(x):
try:
return int(x) if x is not None else None
except Exception:
try:
return int(float(str(x).replace(",", "")))
except Exception:
return None
def _infer_grant_id(obj: dict, fallback_name: str = "") -> Optional[str]:
# 1) explicit fields
for k in ("grant_id","id","competition_id","competitionId"):
if obj.get(k):
return str(obj[k]).replace("competition-","").strip()
# 2) from URL: .../competition/2185/...
url = obj.get("url") or obj.get("source_url") or obj.get("page_url") or ""
m = re.search(r"/competition/(\d+)", url)
if m:
return m.group(1)
# 3) from filename
m2 = re.search(r"competition-(\d+)", fallback_name)
if m2:
return m2.group(1)
return None
def _make_text_blob(title: str, url: str, sections: Dict[str,str]) -> str:
parts = [f"TITLE: {title}", f"URL: {url}"]
for k in ("summary_raw","eligibility_raw","scope_raw","dates_raw","how_to_apply_raw","supporting_information_raw"):
v = sections.get(k)
if v:
parts.append(f"\n[{k}]\n{v}")
# also tolerate alt keys from other crawlers
for k in ("summary","eligibility","scope","dates","how_to_apply","supporting_information"):
v = sections.get(k)
if v and f"[{k}_raw]" not in "".join(parts):
parts.append(f"\n[{k}]\n{v}")
return "\n".join(parts)
def _read_obj(obj: dict, fallback_name: str = "") -> Optional[SupportingDoc]:
gid = _infer_grant_id(obj, fallback_name)
if not gid:
return None
url = obj.get("url") or obj.get("source_url") or obj.get("page_url") or ""
title = (obj.get("title") or obj.get("name") or "").strip()
# Common normalised fields
open_date = obj.get("open_date")
close_date = obj.get("close_date") or obj.get("deadline") or obj.get("closeDate")
notify_date = obj.get("notify_date")
# Funding block: either nested or flat
funding = obj.get("funding") or {}
fmin = _num(funding.get("min") or obj.get("funding_min") or obj.get("min_award") or obj.get("grant_min"))
fmax = _num(funding.get("max") or obj.get("funding_max") or obj.get("max_award") or obj.get("grant_max"))
total_pot = _num(funding.get("total_pot") or obj.get("total_pot") or obj.get("competition_total") or obj.get("total_funding"))
rates = funding.get("rates") if isinstance(funding.get("rates"), str) else obj.get("funding_rates")
# Duration block
dur = obj.get("duration_months") or {}
dmin = _int(dur.get("min") or obj.get("duration_min") or obj.get("project_duration_min_months"))
dmax = _int(dur.get("max") or obj.get("duration_max") or obj.get("project_duration_max_months"))
# Sections: tolerate both nested and flat naming
sections: Dict[str, str] = {}
for k in ("summary_raw","eligibility_raw","scope_raw","dates_raw","how_to_apply_raw","supporting_information_raw",
"summary","eligibility","scope","dates","how_to_apply","supporting_information"):
v = obj.get(k) or (obj.get("sections") or {}).get(k)
if isinstance(v, str) and v.strip():
sections[k] = v
text = _make_text_blob(title, url, sections)
return SupportingDoc(
grant_id=gid,
url=url,
title=title,
open_date=open_date,
close_date=close_date,
notify_date=notify_date,
funding_min=fmin,
funding_max=fmax,
total_pot=total_pot,
funding_rates=rates if isinstance(rates, str) else None,
duration_min=dmin,
duration_max=dmax,
text=text,
sections=sections,
)
def _read_json_file(p: Path) -> Optional[SupportingDoc]:
try:
obj = json.loads(p.read_text(encoding="utf-8"))
return _read_obj(obj, fallback_name=p.name)
except Exception:
return None
def _iter_jsonl(p: Path) -> Iterable[SupportingDoc]:
with p.open("r", encoding="utf-8") as f:
for i, line in enumerate(f, start=1):
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except Exception:
continue
doc = _read_obj(obj, fallback_name=f"{p.name}:{i}")
if doc:
yield doc
def iter_supporting_docs(folder: Path) -> Iterable[SupportingDoc]:
folder = Path(folder)
# Prefer explicit competition-*.json first
found = False
for p in sorted(folder.glob("competition-*.json")):
found = True
doc = _read_json_file(p)
if doc:
yield doc
# Then any *.json
if not found:
for p in sorted(folder.glob("*.json")):
doc = _read_json_file(p)
if doc:
yield doc
# Then *.jsonl (one object per line)
for p in sorted(folder.glob("*.jsonl")):
for doc in _iter_jsonl(p):
yield doc
def load_supporting_docs(folder: Path) -> List[SupportingDoc]:
return list(iter_supporting_docs(folder)) |