| """baron mouse: stratified 50/50 split — train goes to corpus, test held out."""
|
| from __future__ import annotations
|
| from pathlib import Path
|
| import warnings, sys
|
| warnings.filterwarnings("ignore")
|
| import numpy as np, pandas as pd, anndata as ad, scanpy as sc, scipy.sparse as sp
|
| from sklearn.model_selection import train_test_split
|
|
|
| from pathlib import Path as _P_root
|
| ROOT = _P_root(__file__).resolve().parents[2]
|
| ROOT_STR = str(ROOT)
|
| sys.path.insert(0, ROOT_STR)
|
| import importlib.util
|
| spec = importlib.util.spec_from_file_location(
|
| "loaders", f"{ROOT_STR}/panda/data/pancreas_loaders.py")
|
| _mod = importlib.util.module_from_spec(spec); spec.loader.exec_module(_mod)
|
|
|
| HARM = Path(f"{ROOT_STR}/data/corpus/pancreas/harmonized")
|
| HELDOUT = Path(f"{ROOT_STR}/data/corpus/pancreas/held_out_labeled")
|
| HELDOUT.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
| def qc(a, name):
|
| n0 = a.n_obs
|
| sc.pp.filter_cells(a, min_genes=200)
|
| sc.pp.filter_genes(a, min_cells=3)
|
| print(f" [{name}] {n0} -> {a.n_obs} cells, {a.n_vars} genes", flush=True)
|
| return a
|
|
|
|
|
| def main():
|
| print("[baron] loading Baron 2016 GSE84133 …", flush=True)
|
| a = _mod.load_baron()
|
| print(f" raw: {a.shape}", flush=True)
|
|
|
| if "organism" in a.obs.columns:
|
| m_mask = a.obs["organism"] == "mouse"
|
| else:
|
| m_mask = a.obs.index.str.contains("mouse")
|
| a_m = a[m_mask].copy()
|
| print(f" mouse subset: {a_m.shape}", flush=True)
|
| a_m = qc(a_m, "baron_mouse")
|
| a_m.obs["dataset"] = "baron_GSE84133_mouse"
|
| if "assigned_cluster" in a_m.obs.columns:
|
| norm_map = {
|
| "alpha": "alpha", "beta": "beta", "delta": "delta", "gamma": "gamma",
|
| "epsilon": "epsilon", "ductal": "ductal", "acinar": "acinar",
|
| "endothelial": "endothelial", "activated_stellate": "other",
|
| "quiescent_stellate": "other", "schwann": "other",
|
| "mast": "immune", "macrophage": "immune", "t_cell": "immune",
|
| "T_cell": "immune", "b_cell": "immune",
|
| }
|
| a_m.obs["canonical_label"] = a_m.obs["assigned_cluster"].astype(str).str.lower().map(norm_map).fillna("other")
|
| else:
|
| a_m.obs["canonical_label"] = "unknown"
|
|
|
| print(f"[baron] canonical_label counts: "
|
| f"{a_m.obs['canonical_label'].value_counts().to_dict()}", flush=True)
|
|
|
| y = a_m.obs["canonical_label"].astype(str).values
|
| idx = np.arange(a_m.n_obs)
|
| tr_idx, te_idx = train_test_split(idx, test_size=0.5, stratify=y, random_state=42)
|
| a_train = a_m[tr_idx].copy()
|
| a_test = a_m[te_idx].copy()
|
| print(f"[baron] train: {a_train.shape}, test: {a_test.shape}", flush=True)
|
|
|
| for x in (a_train, a_test):
|
| if not sp.issparse(x.X): x.X = sp.csr_matrix(x.X)
|
| x.X = x.X.astype("float32")
|
| x.var_names_make_unique()
|
|
|
| a_train.write_h5ad(HARM / "baron_GSE84133_mouse_train.h5ad", compression="gzip")
|
| a_test.write_h5ad(HELDOUT / "baron_GSE84133_mouse_test.h5ad", compression="gzip")
|
| print(f"[baron] wrote train -> {HARM/'baron_GSE84133_mouse_train.h5ad'} "
|
| f"({(HARM/'baron_GSE84133_mouse_train.h5ad').stat().st_size/1e6:.1f} MB)", flush=True)
|
| print(f"[baron] wrote test -> {HELDOUT/'baron_GSE84133_mouse_test.h5ad'} "
|
| f"({(HELDOUT/'baron_GSE84133_mouse_test.h5ad').stat().st_size/1e6:.1f} MB)", flush=True)
|
|
|
|
|
| if __name__ == "__main__":
|
| main()
|
|
|