"""baron mouse: stratified 50/50 split — train goes to corpus, test held out.""" from __future__ import annotations from pathlib import Path import warnings, sys warnings.filterwarnings("ignore") import numpy as np, pandas as pd, anndata as ad, scanpy as sc, scipy.sparse as sp from sklearn.model_selection import train_test_split from pathlib import Path as _P_root ROOT = _P_root(__file__).resolve().parents[2] ROOT_STR = str(ROOT) sys.path.insert(0, ROOT_STR) import importlib.util spec = importlib.util.spec_from_file_location( "loaders", f"{ROOT_STR}/panda/data/pancreas_loaders.py") _mod = importlib.util.module_from_spec(spec); spec.loader.exec_module(_mod) HARM = Path(f"{ROOT_STR}/data/corpus/pancreas/harmonized") HELDOUT = Path(f"{ROOT_STR}/data/corpus/pancreas/held_out_labeled") HELDOUT.mkdir(parents=True, exist_ok=True) def qc(a, name): n0 = a.n_obs sc.pp.filter_cells(a, min_genes=200) sc.pp.filter_genes(a, min_cells=3) print(f" [{name}] {n0} -> {a.n_obs} cells, {a.n_vars} genes", flush=True) return a def main(): print("[baron] loading Baron 2016 GSE84133 …", flush=True) a = _mod.load_baron() print(f" raw: {a.shape}", flush=True) # mouse subset only (loader sets organism) if "organism" in a.obs.columns: m_mask = a.obs["organism"] == "mouse" else: m_mask = a.obs.index.str.contains("mouse") a_m = a[m_mask].copy() print(f" mouse subset: {a_m.shape}", flush=True) a_m = qc(a_m, "baron_mouse") a_m.obs["dataset"] = "baron_GSE84133_mouse" if "assigned_cluster" in a_m.obs.columns: norm_map = { "alpha": "alpha", "beta": "beta", "delta": "delta", "gamma": "gamma", "epsilon": "epsilon", "ductal": "ductal", "acinar": "acinar", "endothelial": "endothelial", "activated_stellate": "other", "quiescent_stellate": "other", "schwann": "other", "mast": "immune", "macrophage": "immune", "t_cell": "immune", "T_cell": "immune", "b_cell": "immune", } a_m.obs["canonical_label"] = a_m.obs["assigned_cluster"].astype(str).str.lower().map(norm_map).fillna("other") else: a_m.obs["canonical_label"] = "unknown" print(f"[baron] canonical_label counts: " f"{a_m.obs['canonical_label'].value_counts().to_dict()}", flush=True) y = a_m.obs["canonical_label"].astype(str).values idx = np.arange(a_m.n_obs) tr_idx, te_idx = train_test_split(idx, test_size=0.5, stratify=y, random_state=42) a_train = a_m[tr_idx].copy() a_test = a_m[te_idx].copy() print(f"[baron] train: {a_train.shape}, test: {a_test.shape}", flush=True) for x in (a_train, a_test): if not sp.issparse(x.X): x.X = sp.csr_matrix(x.X) x.X = x.X.astype("float32") x.var_names_make_unique() a_train.write_h5ad(HARM / "baron_GSE84133_mouse_train.h5ad", compression="gzip") a_test.write_h5ad(HELDOUT / "baron_GSE84133_mouse_test.h5ad", compression="gzip") print(f"[baron] wrote train -> {HARM/'baron_GSE84133_mouse_train.h5ad'} " f"({(HARM/'baron_GSE84133_mouse_train.h5ad').stat().st_size/1e6:.1f} MB)", flush=True) print(f"[baron] wrote test -> {HELDOUT/'baron_GSE84133_mouse_test.h5ad'} " f"({(HELDOUT/'baron_GSE84133_mouse_test.h5ad').stat().st_size/1e6:.1f} MB)", flush=True) if __name__ == "__main__": main()