kozo2's picture
Upload config.py with huggingface_hub
a58e62a verified
Raw
History Blame Contribute Delete
4.45 kB
"""Per-source-table paths and node metadata, so one set of scripts serves them all.
Every script takes the source table as its first argument (default edge_ML) and
resolves its own files from it:
graph/<src>_filtered_expected_ge5_pyg.pt build_graph.py
<src>_filtered_expected_ge5_n2v.pt node2vec_model.py
figures/umap_coords_<src>_filtered.npy umap_species.py (cache)
figures/umap_species_<src>_filtered_<mode>.png
Node metadata (species) lives in graph.duckdb, split across two tables by id
namespace: MetaboLights ids (MTBLS...) in nodes_ML, Metabolomics Workbench ids
(ST...) in nodes_MW. edge_MLvsMW spans both, so the lookup is their union.
"""
import argparse
import os
import pathlib
SOURCE_TABLES = ("edge_ML", "edge_MLvsMW", "edge_MW_1", "edge_MW_2", "edge_MW_3")
DATA_DIR = pathlib.Path(os.environ.get(
"DATA_DIR", str(pathlib.Path(__file__).resolve().parent.parent)))
DB_PATH = str(DATA_DIR / "edges_filtered.duckdb")
NODE_DB_PATH = str(DATA_DIR / "graph.duckdb") # nodes_ML / nodes_MW live here
NODES_PARQUET = DATA_DIR / "data" / "nodes_expected_ge5.parquet"
EDGE_TABLE = "edges_expected_ge5"
GRAPH_DIR = DATA_DIR / "graph"
FIGURES_DIR = DATA_DIR / "figures"
class Paths:
"""The files belonging to one source table."""
def __init__(self, source_table: str):
if source_table not in SOURCE_TABLES:
raise SystemExit(f"unknown source table {source_table!r}; "
f"expected one of {', '.join(SOURCE_TABLES)}")
self.source_table = source_table
GRAPH_DIR.mkdir(parents=True, exist_ok=True)
FIGURES_DIR.mkdir(parents=True, exist_ok=True)
stem = f"{source_table}_filtered_expected_ge5"
self.graph = str(GRAPH_DIR / f"{stem}_pyg.pt")
self.embeddings = str(DATA_DIR / f"{stem}_n2v.pt")
self.umap_npy = str(FIGURES_DIR / f"umap_coords_{source_table}_filtered.npy")
def figure(self, mode: str, color_by: str = "species") -> str:
return str(FIGURES_DIR /
f"umap_{color_by}_{self.source_table}_filtered_{mode}.png")
def add_source_arg(ap: argparse.ArgumentParser) -> None:
ap.add_argument("--source-table", default="edge_ML", choices=SOURCE_TABLES,
help="which source_table of edges_expected_ge5 to use")
# id namespace -> the database it comes from. MetaboLights study ids start with
# MTBLS, Metabolomics Workbench ids with ST; edge_MLvsMW joins one of each.
DATABASES = (("MTBLS", "MetaboLights (MTBLS)"), ("ST", "Metabolomics Workbench (ST)"))
def database_for(node_ids: list) -> list:
"""Source database per node id, in node_ids order."""
out = []
for i in node_ids:
for prefix, name in DATABASES:
if i.startswith(prefix):
out.append(name)
break
else:
raise SystemExit(f"id {i!r} matches no known database prefix")
return out
def node_source(con) -> str:
"""A SQL relation with the full node property columns, from whatever exists.
The released dataset ships data/nodes_expected_ge5.parquet, which holds the
property rows for exactly the nodes in edges_expected_ge5 — enough for every
graph built from that table, and the only source a downloader has. The
original working tree instead has graph.duckdb, where the rows are split
across nodes_ML (MTBLS ids) and nodes_MW (ST ids).
"""
if NODES_PARQUET.exists():
return f"(SELECT * FROM '{NODES_PARQUET}')"
if pathlib.Path(NODE_DB_PATH).exists():
con.execute(f"ATTACH IF NOT EXISTS '{NODE_DB_PATH}' AS g (READ_ONLY)")
return "(SELECT * FROM g.nodes_ML UNION ALL SELECT * FROM g.nodes_MW)"
raise SystemExit(f"no node properties found: expected {NODES_PARQUET} "
f"or {NODE_DB_PATH}")
def node_property_query(con, column: str = "species") -> str:
"""SQL returning (id, <column>) for every node."""
return f"SELECT id, {column} FROM {node_source(con)}"
def species_for(con, node_ids: list) -> list:
"""species per node id, in node_ids order. Raises if any id is unknown."""
by_id = dict(con.execute(node_property_query(con, "species")).fetchall())
missing = [i for i in node_ids if i not in by_id]
if missing:
raise SystemExit(f"{len(missing):,} nodes have no metadata row, "
f"e.g. {missing[:3]}")
return [by_id[i] for i in node_ids]