Spaces:
Running on Zero
Running on Zero
| """ | |
| Diagnose the confluence count in a saved reach graph: real tributary | |
| junctions, or a data artifact (e.g. the same physical reach digitized | |
| twice, or a tile-boundary duplication in BD TOPO). | |
| Works two ways: | |
| 1. Against your EXISTING risle_nodes.csv/risle_edges.csv (no rerun | |
| needed) -- checks whether each flagged confluence's incoming edges | |
| have suspiciously similar distance_km/elevation_drop_m, which is | |
| the signature of two near-identical tronçons rather than two | |
| genuinely different ones. | |
| 2. Conclusively, once you rerun scripts/build_reach_graphs.py with the | |
| cleabs fix -- if two incoming edges at the same node share the same | |
| cleabs (tronçon ID), that's definitive: same tronçon counted twice, | |
| not two different ones. | |
| Usage: | |
| python -m scripts.diagnose_confluences --data-root datasets --basin risle | |
| """ | |
| import argparse | |
| from pathlib import Path | |
| import pandas as pd | |
| def diagnose(nodes_path: Path, edges_path: Path, n_samples: int = 10) -> None: | |
| nodes_df = pd.read_csv(nodes_path) | |
| edges_df = pd.read_csv(edges_path) | |
| has_is_confluence = "is_confluence" in nodes_df.columns | |
| has_toponym = "toponym" in edges_df.columns | |
| has_cleabs = "cleabs" in edges_df.columns | |
| if has_is_confluence: | |
| confluence_codes = set(nodes_df[nodes_df["is_confluence"]]["station_code"]) | |
| confluences = edges_df[edges_df["target"].isin(confluence_codes)]["target"].value_counts() | |
| raw_in_degree = edges_df["target"].value_counts() | |
| raw_confluences = raw_in_degree[raw_in_degree >= 2] | |
| print(f"{len(confluence_codes)} REAL confluence(s) -- different named river, independent " | |
| f"upstream source, not a split that rejoins downstream -- out of " | |
| f"{len(raw_confluences)} node(s) with raw in-degree >= 2 " | |
| f"({len(nodes_df)} total nodes, {len(edges_df)} edges)") | |
| print(f" The gap ({len(raw_confluences) - len(confluence_codes)} nodes) is same-river " | |
| f"segmentation and/or braided channels that split and rejoin -- both excluded " | |
| f"from the count above, using the authoritative is_confluence column computed " | |
| f"with full graph topology (this script can't re-derive the split-rejoin check " | |
| f"from the flat CSV alone, which is exactly why that column exists).") | |
| elif has_toponym: | |
| name_counts = edges_df.dropna(subset=["toponym"]).groupby("target")["toponym"].nunique() | |
| confluences = name_counts[name_counts > 1] | |
| raw_in_degree = edges_df["target"].value_counts() | |
| raw_confluences = raw_in_degree[raw_in_degree >= 2] | |
| print(f"{len(confluences)} REAL confluence(s) (different named river actually joining), " | |
| f"out of {len(raw_confluences)} node(s) with raw in-degree >= 2 " | |
| f"({len(nodes_df)} total nodes, {len(edges_df)} edges)") | |
| print(f" The gap ({len(raw_confluences) - len(confluences)} nodes) is same-river fine " | |
| f"tronçon segmentation, not real branching -- this is now excluded from the count.") | |
| print(" NOTE: this file predates the split-rejoin fix (no is_confluence column) -- a " | |
| "braided channel with differently-named branches that rejoin downstream would " | |
| "still be miscounted here as a real confluence. Re-run " | |
| "scripts/build_reach_graphs.py for the corrected count.") | |
| else: | |
| in_degree = edges_df["target"].value_counts() | |
| confluences = in_degree[in_degree >= 2] | |
| print(f"{len(confluences)} node(s) with in-degree >= 2, out of {len(nodes_df)} total nodes " | |
| f"and {len(edges_df)} edges") | |
| print() | |
| print("NOTE: this edges.csv doesn't have a 'toponym' column -- it was generated before " | |
| "the name-based confluence fix. Every number above uses the old, less reliable " | |
| "in-degree >= 2 rule. Re-run scripts/build_reach_graphs.py for a corrected count " | |
| "and for this script to give a definitive per-node answer instead of a proxy.") | |
| print() | |
| sample = confluences.head(n_samples) | |
| n_confirmed_duplicate = 0 | |
| n_suspicious = 0 | |
| for node in sample.index: | |
| incoming = edges_df[edges_df["target"] == node] | |
| cols = ["source", "target", "distance_km", "elevation_drop_m"] | |
| cols += [c for c in ("toponym", "cleabs") if c in incoming.columns] | |
| print(f"Node {node}: {len(incoming)} incoming edge(s)") | |
| print(incoming[cols].to_string(index=False)) | |
| if has_toponym: | |
| n_names = incoming["toponym"].nunique() | |
| if n_names > 1: | |
| print(f" -> REAL: {n_names} distinct named river(s) join here.") | |
| else: | |
| print(f" -> Same toponym on every incoming edge -- not counted as a real " | |
| f"confluence (this shouldn't appear in the sample above; if it does, " | |
| f"something's inconsistent between this script and build_reach_graph.py).") | |
| elif has_cleabs and incoming["cleabs"].notna().all(): | |
| if incoming["cleabs"].nunique() < len(incoming): | |
| print(" -> CONFIRMED DUPLICATE: two+ incoming edges share the same cleabs " | |
| "(same physical tronçon counted more than once).") | |
| n_confirmed_duplicate += 1 | |
| else: | |
| dists = incoming["distance_km"].dropna() | |
| if len(dists) >= 2 and (dists.max() - dists.min()) < 0.05: | |
| print(" -> SUSPICIOUS: incoming edges have near-identical distance_km " | |
| "(within 50m) -- consistent with the same reach being counted twice, " | |
| "not two distinct tributaries joining.") | |
| n_suspicious += 1 | |
| print() | |
| print("=" * 60) | |
| if has_toponym: | |
| print(f"All {len(sample)} sampled nodes have >1 distinct named river joining -- " | |
| f"that's the definition being used, so this should always be 100%. If any " | |
| f"printed 'Same toponym' above, report it as a bug.") | |
| elif has_cleabs: | |
| print(f"Of {len(sample)} sampled confluences: {n_confirmed_duplicate} confirmed " | |
| f"duplicate-tronçon artifacts, {len(sample) - n_confirmed_duplicate} have " | |
| f"genuinely distinct incoming tronçons (real confluence candidates).") | |
| else: | |
| print(f"Of {len(sample)} sampled confluences: {n_suspicious} look suspicious " | |
| f"(near-identical incoming edge distances). Re-run scripts/build_reach_graphs.py " | |
| f"for a conclusive, name-based answer instead of this proxy signal.") | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description="Diagnose reach-graph confluence counts") | |
| parser.add_argument("--data-root", type=Path, default=Path("datasets")) | |
| parser.add_argument("--basin", choices=["eure", "risle"], required=True) | |
| parser.add_argument("--n-samples", type=int, default=10) | |
| args = parser.parse_args() | |
| graph_dir = args.data_root / "reach_graph" | |
| diagnose(graph_dir / f"{args.basin}_nodes.csv", graph_dir / f"{args.basin}_edges.csv", | |
| n_samples=args.n_samples) | |
| if __name__ == "__main__": | |
| main() |