File size: 4,321 Bytes
a74054f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
"""
Download catchment (drainage basin) area for all stations from Hub'Eau's
own hydrometrie referentiel — the same API download_hubeau.py already
uses, just a different endpoint.

Catchment area is NOT on the station referentiel directly; it lives on
the SITE referentiel (`surface_bv`, "Superficie du BV en km2"). A site
can have multiple stations, so this does two passes:
  1. referentiel/stations -> get each station's code_site
  2. referentiel/sites     -> get surface_bv for those code_site values
  3. join back onto station_code

This gives real drainage-area data usable for area-ratio discharge
scaling between gauges on the same river (Q_ungauged ~= Q_gauged *
(area_ungauged / area_gauged)) — a physically grounded alternative to
naive geographic proximity interpolation.
"""
import requests
import pandas as pd
from pathlib import Path

BASE_URL = "https://hubeau.eaufrance.fr/api/v2/hydrometrie"
OUTPUT_DIR = Path("datasets")
OUTPUT_DIR.mkdir(exist_ok=True)

stations_df = pd.read_csv("datasets/station_list.csv")
station_codes = stations_df["station_code"].tolist()

print(f"Fetching catchment area for {len(station_codes)} stations...")
print()


def fetch_all(endpoint: str, params: dict) -> list:
    """Fetch every page from a Hub'Eau referentiel endpoint (cursor pagination)."""
    all_data = []
    cursor = ""
    while True:
        query = dict(params, cursor=cursor, size=1000, format="json")
        response = requests.get(f"{BASE_URL}/{endpoint}", params=query, timeout=30)
        if response.status_code not in (200, 206):
            print(f"  Error {response.status_code}: {response.text[:200]}")
            break
        data = response.json()
        results = data.get("data", [])
        if not results:
            break
        all_data.extend(results)
        next_url = data.get("next")
        if not next_url:
            break
        if "cursor=" in next_url:
            cursor = next_url.split("cursor=")[1].split("&")[0]
        else:
            break
    return all_data


# --- Step 1: station -> code_site -----------------------------------------
print("=" * 80)
print("STEP 1: Looking up code_site for each station")
print("=" * 80)

station_to_site = {}
# code_station accepts up to 100 comma-separated codes per request
for i in range(0, len(station_codes), 100):
    batch = station_codes[i:i + 100]
    results = fetch_all("referentiel/stations", {"code_station": ",".join(batch)})
    for r in results:
        station_to_site[r["code_station"]] = r.get("code_site")
        print(f"  {r['code_station']} -> site {r.get('code_site')}")

missing = set(station_codes) - set(station_to_site)
if missing:
    print(f"\n  Warning: {len(missing)} station(s) not found in referentiel/stations: {missing}")

# --- Step 2: code_site -> surface_bv ---------------------------------------
print()
print("=" * 80)
print("STEP 2: Fetching surface_bv (catchment area) for each site")
print("=" * 80)

site_codes = sorted(set(s for s in station_to_site.values() if s))
site_to_area = {}
for i in range(0, len(site_codes), 100):
    batch = site_codes[i:i + 100]
    results = fetch_all("referentiel/sites", {"code_site": ",".join(batch)})
    for r in results:
        area = r.get("surface_bv")
        site_to_area[r["code_site"]] = area
        print(f"  site {r['code_site']} ({r.get('libelle_site', '?')}): "
              f"{area if area is not None else 'no data'} km²")

# --- Step 3: join back onto station_code -----------------------------------
rows = []
for code in station_codes:
    site = station_to_site.get(code)
    area = site_to_area.get(site) if site else None
    rows.append({"station_code": code, "code_site": site, "catchment_area_km2": area})

result_df = pd.DataFrame(rows)
output_file = OUTPUT_DIR / "catchment_area.csv"
result_df.to_csv(output_file, index=False)

print()
print("=" * 80)
print("DONE")
print("=" * 80)
n_ok = result_df["catchment_area_km2"].notna().sum()
print(f"Catchment area found for {n_ok}/{len(result_df)} stations")
print(f"Saved to: {output_file}")
if n_ok < len(result_df):
    missing_codes = result_df[result_df["catchment_area_km2"].isna()]["station_code"].tolist()
    print(f"\nNo area found for: {missing_codes}")
    print("(Hub'Eau doesn't publish surface_bv for every site — some are genuinely blank.)")