""" Diagnose the confluence count in a saved reach graph: real tributary junctions, or a data artifact (e.g. the same physical reach digitized twice, or a tile-boundary duplication in BD TOPO). Works two ways: 1. Against your EXISTING risle_nodes.csv/risle_edges.csv (no rerun needed) -- checks whether each flagged confluence's incoming edges have suspiciously similar distance_km/elevation_drop_m, which is the signature of two near-identical tronçons rather than two genuinely different ones. 2. Conclusively, once you rerun scripts/build_reach_graphs.py with the cleabs fix -- if two incoming edges at the same node share the same cleabs (tronçon ID), that's definitive: same tronçon counted twice, not two different ones. Usage: python -m scripts.diagnose_confluences --data-root datasets --basin risle """ import argparse from pathlib import Path import pandas as pd def diagnose(nodes_path: Path, edges_path: Path, n_samples: int = 10) -> None: nodes_df = pd.read_csv(nodes_path) edges_df = pd.read_csv(edges_path) has_is_confluence = "is_confluence" in nodes_df.columns has_toponym = "toponym" in edges_df.columns has_cleabs = "cleabs" in edges_df.columns if has_is_confluence: confluence_codes = set(nodes_df[nodes_df["is_confluence"]]["station_code"]) confluences = edges_df[edges_df["target"].isin(confluence_codes)]["target"].value_counts() raw_in_degree = edges_df["target"].value_counts() raw_confluences = raw_in_degree[raw_in_degree >= 2] print(f"{len(confluence_codes)} REAL confluence(s) -- different named river, independent " f"upstream source, not a split that rejoins downstream -- out of " f"{len(raw_confluences)} node(s) with raw in-degree >= 2 " f"({len(nodes_df)} total nodes, {len(edges_df)} edges)") print(f" The gap ({len(raw_confluences) - len(confluence_codes)} nodes) is same-river " f"segmentation and/or braided channels that split and rejoin -- both excluded " f"from the count above, using the authoritative is_confluence column computed " f"with full graph topology (this script can't re-derive the split-rejoin check " f"from the flat CSV alone, which is exactly why that column exists).") elif has_toponym: name_counts = edges_df.dropna(subset=["toponym"]).groupby("target")["toponym"].nunique() confluences = name_counts[name_counts > 1] raw_in_degree = edges_df["target"].value_counts() raw_confluences = raw_in_degree[raw_in_degree >= 2] print(f"{len(confluences)} REAL confluence(s) (different named river actually joining), " f"out of {len(raw_confluences)} node(s) with raw in-degree >= 2 " f"({len(nodes_df)} total nodes, {len(edges_df)} edges)") print(f" The gap ({len(raw_confluences) - len(confluences)} nodes) is same-river fine " f"tronçon segmentation, not real branching -- this is now excluded from the count.") print(" NOTE: this file predates the split-rejoin fix (no is_confluence column) -- a " "braided channel with differently-named branches that rejoin downstream would " "still be miscounted here as a real confluence. Re-run " "scripts/build_reach_graphs.py for the corrected count.") else: in_degree = edges_df["target"].value_counts() confluences = in_degree[in_degree >= 2] print(f"{len(confluences)} node(s) with in-degree >= 2, out of {len(nodes_df)} total nodes " f"and {len(edges_df)} edges") print() print("NOTE: this edges.csv doesn't have a 'toponym' column -- it was generated before " "the name-based confluence fix. Every number above uses the old, less reliable " "in-degree >= 2 rule. Re-run scripts/build_reach_graphs.py for a corrected count " "and for this script to give a definitive per-node answer instead of a proxy.") print() sample = confluences.head(n_samples) n_confirmed_duplicate = 0 n_suspicious = 0 for node in sample.index: incoming = edges_df[edges_df["target"] == node] cols = ["source", "target", "distance_km", "elevation_drop_m"] cols += [c for c in ("toponym", "cleabs") if c in incoming.columns] print(f"Node {node}: {len(incoming)} incoming edge(s)") print(incoming[cols].to_string(index=False)) if has_toponym: n_names = incoming["toponym"].nunique() if n_names > 1: print(f" -> REAL: {n_names} distinct named river(s) join here.") else: print(f" -> Same toponym on every incoming edge -- not counted as a real " f"confluence (this shouldn't appear in the sample above; if it does, " f"something's inconsistent between this script and build_reach_graph.py).") elif has_cleabs and incoming["cleabs"].notna().all(): if incoming["cleabs"].nunique() < len(incoming): print(" -> CONFIRMED DUPLICATE: two+ incoming edges share the same cleabs " "(same physical tronçon counted more than once).") n_confirmed_duplicate += 1 else: dists = incoming["distance_km"].dropna() if len(dists) >= 2 and (dists.max() - dists.min()) < 0.05: print(" -> SUSPICIOUS: incoming edges have near-identical distance_km " "(within 50m) -- consistent with the same reach being counted twice, " "not two distinct tributaries joining.") n_suspicious += 1 print() print("=" * 60) if has_toponym: print(f"All {len(sample)} sampled nodes have >1 distinct named river joining -- " f"that's the definition being used, so this should always be 100%. If any " f"printed 'Same toponym' above, report it as a bug.") elif has_cleabs: print(f"Of {len(sample)} sampled confluences: {n_confirmed_duplicate} confirmed " f"duplicate-tronçon artifacts, {len(sample) - n_confirmed_duplicate} have " f"genuinely distinct incoming tronçons (real confluence candidates).") else: print(f"Of {len(sample)} sampled confluences: {n_suspicious} look suspicious " f"(near-identical incoming edge distances). Re-run scripts/build_reach_graphs.py " f"for a conclusive, name-based answer instead of this proxy signal.") def main() -> None: parser = argparse.ArgumentParser(description="Diagnose reach-graph confluence counts") parser.add_argument("--data-root", type=Path, default=Path("datasets")) parser.add_argument("--basin", choices=["eure", "risle"], required=True) parser.add_argument("--n-samples", type=int, default=10) args = parser.parse_args() graph_dir = args.data_root / "reach_graph" diagnose(graph_dir / f"{args.basin}_nodes.csv", graph_dir / f"{args.basin}_edges.csv", n_samples=args.n_samples) if __name__ == "__main__": main()