File size: 5,909 Bytes
19de729
 
 
 
 
 
 
bfcc872
 
19de729
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
bfcc872
 
19de729
bfcc872
19de729
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
# src/analyzer/data_loader_supporting.py
from __future__ import annotations
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, List, Optional, Iterable, Union
import json, re

from .utils.text import to_number

@dataclass
class SupportingDoc:
    grant_id: str
    url: str
    title: str
    open_date: Optional[str]
    close_date: Optional[str]
    notify_date: Optional[str]
    funding_min: Optional[float]
    funding_max: Optional[float]
    total_pot: Optional[float]
    funding_rates: Optional[str]
    duration_min: Optional[int]
    duration_max: Optional[int]
    text: str                    # flattened blob for retrieval
    sections: Dict[str, str]     # raw sections, if present

# Number parsing moved to utils.text.to_number()
# Keeping wrapper for backward compatibility
def _num(x):
    return to_number(x)

def _int(x):
    try:
        return int(x) if x is not None else None
    except Exception:
        try:
            return int(float(str(x).replace(",", "")))
        except Exception:
            return None

def _infer_grant_id(obj: dict, fallback_name: str = "") -> Optional[str]:
    # 1) explicit fields
    for k in ("grant_id","id","competition_id","competitionId"):
        if obj.get(k):
            return str(obj[k]).replace("competition-","").strip()
    # 2) from URL: .../competition/2185/...
    url = obj.get("url") or obj.get("source_url") or obj.get("page_url") or ""
    m = re.search(r"/competition/(\d+)", url)
    if m:
        return m.group(1)
    # 3) from filename
    m2 = re.search(r"competition-(\d+)", fallback_name)
    if m2:
        return m2.group(1)
    return None

def _make_text_blob(title: str, url: str, sections: Dict[str,str]) -> str:
    parts = [f"TITLE: {title}", f"URL: {url}"]
    for k in ("summary_raw","eligibility_raw","scope_raw","dates_raw","how_to_apply_raw","supporting_information_raw"):
        v = sections.get(k)
        if v:
            parts.append(f"\n[{k}]\n{v}")
    # also tolerate alt keys from other crawlers
    for k in ("summary","eligibility","scope","dates","how_to_apply","supporting_information"):
        v = sections.get(k)
        if v and f"[{k}_raw]" not in "".join(parts):
            parts.append(f"\n[{k}]\n{v}")
    return "\n".join(parts)

def _read_obj(obj: dict, fallback_name: str = "") -> Optional[SupportingDoc]:
    gid = _infer_grant_id(obj, fallback_name)
    if not gid:
        return None

    url = obj.get("url") or obj.get("source_url") or obj.get("page_url") or ""
    title = (obj.get("title") or obj.get("name") or "").strip()

    # Common normalised fields
    open_date  = obj.get("open_date")
    close_date = obj.get("close_date") or obj.get("deadline") or obj.get("closeDate")
    notify_date = obj.get("notify_date")

    # Funding block: either nested or flat
    funding = obj.get("funding") or {}
    fmin = _num(funding.get("min") or obj.get("funding_min") or obj.get("min_award") or obj.get("grant_min"))
    fmax = _num(funding.get("max") or obj.get("funding_max") or obj.get("max_award") or obj.get("grant_max"))
    total_pot = _num(funding.get("total_pot") or obj.get("total_pot") or obj.get("competition_total") or obj.get("total_funding"))
    rates = funding.get("rates") if isinstance(funding.get("rates"), str) else obj.get("funding_rates")

    # Duration block
    dur = obj.get("duration_months") or {}
    dmin = _int(dur.get("min") or obj.get("duration_min") or obj.get("project_duration_min_months"))
    dmax = _int(dur.get("max") or obj.get("duration_max") or obj.get("project_duration_max_months"))

    # Sections: tolerate both nested and flat naming
    sections: Dict[str, str] = {}
    for k in ("summary_raw","eligibility_raw","scope_raw","dates_raw","how_to_apply_raw","supporting_information_raw",
              "summary","eligibility","scope","dates","how_to_apply","supporting_information"):
        v = obj.get(k) or (obj.get("sections") or {}).get(k)
        if isinstance(v, str) and v.strip():
            sections[k] = v

    text = _make_text_blob(title, url, sections)
    return SupportingDoc(
        grant_id=gid,
        url=url,
        title=title,
        open_date=open_date,
        close_date=close_date,
        notify_date=notify_date,
        funding_min=fmin,
        funding_max=fmax,
        total_pot=total_pot,
        funding_rates=rates if isinstance(rates, str) else None,
        duration_min=dmin,
        duration_max=dmax,
        text=text,
        sections=sections,
    )

def _read_json_file(p: Path) -> Optional[SupportingDoc]:
    try:
        obj = json.loads(p.read_text(encoding="utf-8"))
        return _read_obj(obj, fallback_name=p.name)
    except Exception:
        return None

def _iter_jsonl(p: Path) -> Iterable[SupportingDoc]:
    with p.open("r", encoding="utf-8") as f:
        for i, line in enumerate(f, start=1):
            line = line.strip()
            if not line:
                continue
            try:
                obj = json.loads(line)
            except Exception:
                continue
            doc = _read_obj(obj, fallback_name=f"{p.name}:{i}")
            if doc:
                yield doc

def iter_supporting_docs(folder: Path) -> Iterable[SupportingDoc]:
    folder = Path(folder)
    # Prefer explicit competition-*.json first
    found = False
    for p in sorted(folder.glob("competition-*.json")):
        found = True
        doc = _read_json_file(p)
        if doc:
            yield doc
    # Then any *.json
    if not found:
        for p in sorted(folder.glob("*.json")):
            doc = _read_json_file(p)
            if doc:
                yield doc
    # Then *.jsonl (one object per line)
    for p in sorted(folder.glob("*.jsonl")):
        for doc in _iter_jsonl(p):
            yield doc

def load_supporting_docs(folder: Path) -> List[SupportingDoc]:
    return list(iter_supporting_docs(folder))