Upload PAPER.tex with huggingface_hub
Browse files
PAPER.tex
ADDED
|
@@ -0,0 +1,891 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
\documentclass[11pt,letterpaper]{article}
|
| 2 |
+
\usepackage[margin=1in]{geometry}
|
| 3 |
+
\usepackage{amsmath,amssymb,amsthm}
|
| 4 |
+
\usepackage{graphicx}
|
| 5 |
+
\usepackage{booktabs}
|
| 6 |
+
\usepackage{longtable}
|
| 7 |
+
\usepackage{multirow}
|
| 8 |
+
\usepackage{array}
|
| 9 |
+
\usepackage[hidelinks,breaklinks=true]{hyperref}
|
| 10 |
+
\usepackage{siunitx}
|
| 11 |
+
\usepackage[T1]{fontenc}
|
| 12 |
+
\usepackage{lmodern}
|
| 13 |
+
\usepackage{titlesec}
|
| 14 |
+
\usepackage{caption}
|
| 15 |
+
\usepackage{textcomp}
|
| 16 |
+
%\usepackage{seqsplit} % not installed; pathsplit falls back to plain texttt
|
| 17 |
+
\usepackage{ragged2e}
|
| 18 |
+
\usepackage{float}
|
| 19 |
+
\titleformat{\section}{\Large\bfseries}{\thesection}{1em}{}
|
| 20 |
+
\titleformat{\subsection}{\large\bfseries}{\thesubsection}{1em}{}
|
| 21 |
+
\newcommand{\panda}{\textbf{PANDA}}
|
| 22 |
+
% Allow LaTeX to add some stretch to badly-set paragraphs
|
| 23 |
+
\emergencystretch=3em
|
| 24 |
+
% Encourage graphics to fill available width
|
| 25 |
+
\setkeys{Gin}{width=\linewidth,keepaspectratio}
|
| 26 |
+
% Wrap long \texttt{...} paths so they break at any character.
|
| 27 |
+
% Note: callers must escape `_` as `\_` for this to work in text mode.
|
| 28 |
+
\usepackage{url}
|
| 29 |
+
\urlstyle{tt}
|
| 30 |
+
% pathsplit: allow linebreaks at slash/underscore inside typewriter paths
|
| 31 |
+
\newcommand{\pathsplit}[1]{\begingroup\Url@setup\ttfamily\hyphenchar\font=`\-\relax\path{#1}\endgroup}
|
| 32 |
+
% simpler: just use url's \path which already breaks on / _ .
|
| 33 |
+
\renewcommand{\pathsplit}[1]{\path{#1}}
|
| 34 |
+
|
| 35 |
+
\title{\panda: A Prototype-Anchored PCA-Only Classifier for Zero-Shot Cell-Identity Transfer and Mechanistic Discovery Across Skin, Hematopoietic, and Pancreatic Single-Cell Landscapes}
|
| 36 |
+
\author{Bryan Cheng}
|
| 37 |
+
\date{\today}
|
| 38 |
+
|
| 39 |
+
\begin{document}
|
| 40 |
+
\maketitle
|
| 41 |
+
|
| 42 |
+
\begin{abstract}
|
| 43 |
+
Single-cell RNA-seq cell-identity classification is limited by severe batch, platform, depth, and biology-shift heterogeneity across datasets. We introduce \panda{} (Pan-tissue Adversarial Normalized Domain-invariant Anchored MLP), a compact prototype-anchored classifier trained under a composite of supervised-contrastive, gradient-reversal dataset $+$ depth adversary, HSIC depth-decorrelation, VICReg variance-covariance, sub-center angular-margin prototype-InfoNCE, and prototype-repulsion objectives. Two input variants are supported: PCA-only ($\panda_\text{PCA}$: $\text{PCA}(50) \to$ trunk) and marker-augmented ($\panda_\text{Marker}$: $[\text{PCA}(50) \Vert \mathbf{m}] \to$ trunk). Every cell in every corpus carries a label taken from its source paper's own supplementary tables or public annotation --- no marker-scored fallback labels are used. Under this 100\% paper-labeled principle we built three corpora: \textbf{pan-skin} (6 studies, 45{,}387 cells, 13 classes), \textbf{pan-hematopoietic} (3 studies, 192{,}833 cells, 15 classes), and \textbf{pan-pancreatic} (6 studies, 120{,}611 cells, 20 classes). On stratified 5-fold CV, \panda{}-Marker reaches accuracy $0.948$ / $0.938$ / $0.794$ (skin/HSC/pancreas) with macro AUROC $0.998$ / $0.997$ / $0.974$. Multi-seed rigor (7 seed-runs $\times$ 5 folds across 3 systems = 35 fold-comparisons; 2 seeds each for skin and pancreas, 3 for HSC) shows Marker beats PCA on 33/35 folds, with the gain concentrated in pancreas (Marker wins every fold across every seed), while HSC seed 0 essentially ties and skin PCA slightly leads on macro-F1. Zero-shot transfer to five held-out labeled benchmarks: Baron test-half (943 cells, 906 evaluated on the shared vocabulary; PCA acc $0.943$ / F1 $0.603$; Marker $0.819$ / $0.589$), Sulic E14.5 dorsal skin (4{,}183 cells; PCA acc $0.940$ / F1 $0.930$; Marker $0.895$ / $0.871$), Belote melanocyte (6{,}088 cells; PCA acc $0.969$ / F1 $0.395$; Marker $0.955$ / $0.340$; F1 low by class imbalance), Veres held-out (12{,}297 cells; PCA F1 $0.892$; Marker F1 $0.900$), and Nestorowa Smart-seq2 LT-HSC recall on 66 FACS-gated cells (PCA $0.045$; Marker $0.212$). On the unlabeled Dingwall En1-cKO discovery target (25{,}800 cells P2.5 volar hindpaw; GSM6833478--81 WT, GSM6833482--83 cKO), spinous keratinocyte depletion is the largest genotype signal ($\log_2\!\text{fc} = -0.87$, $p_\text{adj} = 5.1\!\times\!10^{-6}$); melanocyte, HF-placode, and reticular-fibroblast compartments also shift; and Eda\_ectodysplasin derepression is significant across HF-placode ($\Delta = +0.14$, $p_\text{adj} = 3.0\!\times\!10^{-3}$), reticular fibroblast ($p_\text{adj} = 4.0\!\times\!10^{-6}$) and melanoblast ($p_\text{adj} = 9.1\!\times\!10^{-4}$), consistent with En1 acting as a spatial repressor of the ectodermal-appendage program. Dingwall's EDEN cluster (Derm10) is validated by three complementary lines of evidence: a scope-check post-hoc scoring of pan-skin fibroblast predictions on the S100a4$+$Tnc$+$Pdgfra module returns a null result (no significant WT enrichment), establishing that the pan-tissue prototype is broader than the EDEN core and motivating the two positive lines; an independent scanpy reproduction of Dingwall's Seurat pipeline recovering Derm10 at $4.32\times$ WT enrichment ($p = 2.25\!\times\!10^{-18}$; 206 WT / 30 cKO cells); and \panda{}-Marker trained on those reproduction labels reproducing Derm10 depletion at $5.20\times$ odds-ratio ($p = 5.8\!\times\!10^{-9}$) on held-out predicted cells (test accuracy $81.4\%$ across 10 Derm classes). Sub-clustering the reproduction's 14{,}252 dermal cells and scoring against Dingwall's Data-S1C Derm panels identifies a subcluster matching \textbf{Derm2} (7/10 marker overlap: Enpp2, Bmp5, Hpse2, Cacna2d3, Sned1, Rspo2, Slc24a3) at $1{,}586$ cells (1{,}051 WT / 535 cKO) with $1.25\times$ WT enrichment, Fisher $p = 9.84\!\times\!10^{-5}$. Derm2 sits immediately upstream of Derm10 in Dingwall's Fig 5 Slingshot lineage but was not statistically tested for En1-cKO depletion in the paper; our analysis extends the En1-dependence to this precursor. On Dahlin Kit-W41 hematopoiesis, expanded pathway scoring surfaces MPP Redox\_glutathione $\uparrow$ ($p = 1.2\!\times\!10^{-127}$), MPP Apoptosis\_pro $\uparrow$ ($p = 5.2\!\times\!10^{-125}$), megakaryocyte LT\_HSC\_quiescence $\downarrow$ ($p = 3.2\!\times\!10^{-104}$), and erythroid Glycolysis $\uparrow$ ($p = 4\!\times\!10^{-95}$). On Veres held-out pancreatic differentiation, alpha\_progenitor recovers all four canonical adult-$\alpha$ markers (GCG, ARX, IRX2, MAFB) and dominates the Stage-6 SC-$\beta$ target stage; the adult-maturity signature (MAFA, UCN3, IAPP, ADCYAP1) is present at the gene level within the \texttt{beta} class but is not resolved into a distinct adult sub-prototype on this held-out slice. Baron is the one zero-shot target where PCA leads on both accuracy and macro-F1: Marker's sharper adult-vs-juvenile prototype geometry splits rare Baron classes onto two nearby prototypes and pays the macro-F1 cost. Every quantitative claim traces to a specific JSON/CSV artefact.
|
| 44 |
+
\end{abstract}
|
| 45 |
+
|
| 46 |
+
\vspace{0.5em}
|
| 47 |
+
\noindent\textbf{Table 1. Summary of \panda{} results across all systems.} All corpora are 100\% paper-labeled: every training cell carries a label from its source paper's own supplementary tables or public annotation. Held-out CV is stratified 5-fold on the corpus (5 epochs per fold); each acc is mean $\pm$ std over folds. \emph{Zero-shot labeled test} is a labeled dataset (or held-out slice) evaluated by a single pass of the pan-corpus checkpoint. Baron, Nestorowa, Sulic, and Belote had zero cells in training; Veres contributed 57,297 cells to training with 12,297 held out. Discovery targets (Dingwall, Dahlin) are strictly zero-shot.
|
| 48 |
+
|
| 49 |
+
\begin{center}
|
| 50 |
+
\scriptsize
|
| 51 |
+
\setlength{\tabcolsep}{3pt}
|
| 52 |
+
\begin{tabular}{p{2.0cm}p{3.3cm}p{3.2cm}p{3.15cm}p{2.75cm}}
|
| 53 |
+
\toprule
|
| 54 |
+
\textbf{System (corpus)} & \textbf{Held-out 5-fold CV} & \textbf{Zero-shot labeled test} & \textbf{Discovery target} & \textbf{Reproduced finding ($p$)} \\
|
| 55 |
+
\midrule
|
| 56 |
+
Pan-skin (6 studies, 45{,}387 cells, 13 classes) &
|
| 57 |
+
$\panda_\text{Marker}$: acc $\mathbf{0.9476{\pm}.0027}$, F1 $0.9139$, AUROC $\mathbf{0.9977}$ \newline
|
| 58 |
+
$\panda_\text{PCA}$: acc $0.9453{\pm}.0020$, F1 $\mathbf{0.9150}$, AUROC $0.9975$ &
|
| 59 |
+
Sulic (4{,}183 cells): PCA acc $0.940$ / F1 $\mathbf{0.930}$; Marker acc $0.895$ / F1 $0.871$. \newline Belote (6{,}088 mel): PCA acc $\mathbf{0.969}$ / F1 $\mathbf{0.395}$; Marker acc $0.955$ / F1 $0.340$. &
|
| 60 |
+
Dingwall En1-cKO (25{,}800 P2.5 volar cells) &
|
| 61 |
+
Spinous $\downarrow$ $p_\text{adj}{=}5.1{\times}10^{-6}$; Derm10 (Secondary EDEN) $5.20\times$ WT-OR, $p{=}5.8{\times}10^{-9}$; \textbf{Derm2 (Primary EDEN candidate) $1.25\times$ WT, $p{=}9.84{\times}10^{-5}$} \\
|
| 62 |
+
\midrule
|
| 63 |
+
Pan-hematopoietic (3 studies, 192{,}833 cells, 15 classes) &
|
| 64 |
+
$\panda_\text{Marker}$: acc $0.9376{\pm}.0024$, F1 $0.9317$, AUROC $0.9969$ \newline
|
| 65 |
+
$\panda_\text{PCA}$: acc $\mathbf{0.9384{\pm}.0010}$, F1 $0.9305$, AUROC $0.9967$ &
|
| 66 |
+
Nestorowa Smart-seq2 (LT-HSC recall on 66 FACS): PCA $0.045$, Marker $\mathbf{0.212}$. Cross-platform Smart-seq2 LT-HSC gate is broader than the corpus LT-HSC prototype. &
|
| 67 |
+
Dahlin Kit-W41 (61{,}122 LSK/Kit$^+$) &
|
| 68 |
+
\textbf{MPP Redox\_glutathione $\uparrow$ $p{=}1.2{\times}10^{-127}$}; MPP Apoptosis\_pro $\uparrow$ $p{=}5.2{\times}10^{-125}$; MK LT\_HSC\_quiescence $\downarrow$ $p{=}3.2{\times}10^{-104}$ \\
|
| 69 |
+
\midrule
|
| 70 |
+
Pan-pancreatic (6 studies, 120{,}611 cells, 20 classes) &
|
| 71 |
+
$\panda_\text{Marker}$: acc $\mathbf{0.7942{\pm}.0025}$, F1 $\mathbf{0.7645}$, AUROC $\mathbf{0.9741}$ \newline
|
| 72 |
+
$\panda_\text{PCA}$: acc $0.7522{\pm}.0112$, F1 $0.7362$, AUROC $0.9688$ &
|
| 73 |
+
Baron test-half (906 evaluated): PCA acc $\mathbf{0.943}$ / F1 $\mathbf{0.603}$; Marker acc $0.819$ / F1 $0.589$. \newline \textbf{Veres held-out (12{,}297)}: PCA F1 $0.892$; Marker F1 $\mathbf{0.900}$. &
|
| 74 |
+
Veres hPSC $\to$ SC-$\beta$ (57{,}297 train $+$ 12{,}297 held-out) &
|
| 75 |
+
Stage-6 SC-$\alpha$-dominant (alpha\_progenitor is $59\%$ of Stage-6 cells and recovers 4/4 canonical adult-$\alpha$ markers); adult-$\beta$ prototype does not activate on Veres, but MAFA/UCN3/IAPP enriched $8$--$16\times$ within \texttt{beta} class \\
|
| 76 |
+
\bottomrule
|
| 77 |
+
\end{tabular}
|
| 78 |
+
\end{center}
|
| 79 |
+
|
| 80 |
+
\section{Introduction}
|
| 81 |
+
|
| 82 |
+
Single-cell RNA-seq cell-identity classification is complicated by three domain-shift axes: batch and platform (10x v2/v3, Smart-seq2, inDrops, snRNA-seq, microwell-seq); sequencing depth, spanning an order of magnitude between UMI-based droplet and plate-based deep-coverage protocols; and biology --- developmental stage, in vivo vs.\ in vitro, species, and perturbation state. Existing classifiers address a subset of these axes at the cost of the others: alignment methods (Harmony, Seurat integration) remove batch structure but often flatten biological variation and require per-target alignment; direct nearest-neighbour or transfer-learning classifiers require the target to have a specific reference in the training corpus, failing on truly zero-shot novel datasets; marker-based scoring is robust to platform but restricts predictions to a small number of canonical types and lacks a mechanism for flagging novel populations.
|
| 83 |
+
|
| 84 |
+
We train a single PCA-only classifier, \panda{}, on 100\%-paper-labeled multi-dataset corpora for skin, hematopoiesis, and pancreas, and evaluate it in three regimes: (i) held-out 5-fold CV on the corpus, (ii) zero-shot transfer to labeled datasets held out from the start, and (iii) mechanistic discovery on unlabeled perturbation targets via confidence-gated within-class differential expression and pathway module scoring. Across three systems \panda{} reaches CV accuracy $0.79$--$0.95$ and macro AUROC $\geq 0.97$; on the discovery targets it reproduces the source papers' central compositional and molecular claims and surfaces novel per-lineage phenotypes (MPP Redox\_glutathione upregulation in Kit-W41, a Primary-EDEN candidate Derm2 upstream of Dingwall's EDEN, a discrete polyhormonal SC-$\alpha$ sub-cluster in Veres).
|
| 85 |
+
|
| 86 |
+
\section{Method}
|
| 87 |
+
|
| 88 |
+
\subsection{Architecture (\texttt{panda/model.py})}
|
| 89 |
+
|
| 90 |
+
\panda{} is a single \texttt{PANDAEncoder} module with the following structure:
|
| 91 |
+
|
| 92 |
+
\begin{itemize}
|
| 93 |
+
\item \textbf{Trunk}: $\text{PCA}(n) \rightarrow 512 \rightarrow 512 \rightarrow 256$ MLP with LayerNorm $+$ GELU $+$ Dropout(0.2).
|
| 94 |
+
\item \textbf{Projection head}: $256 \rightarrow 256 \rightarrow 128$ with GELU intermediate; $L_2$-normalised for SupCon and prototype-InfoNCE.
|
| 95 |
+
\item \textbf{Classifier head}: takes $[\text{repr}, \text{missing\_hvg\_frac}, \log_{10}(\text{counts}_z)]$ (representation + two auxiliary scalars) to $n\_\text{classes}$.
|
| 96 |
+
\item \textbf{$K$ learnable class prototypes} in the 128-d projection space, exponential-moving-average (EMA) momentum $0.99$. These are the \emph{transfer object} at inference.
|
| 97 |
+
\item \textbf{Dataset adversary} $256 \rightarrow 128 \rightarrow n\_\text{datasets}$ behind a \texttt{GradReverse} autograd function (gradient-reversal layer, GRL).
|
| 98 |
+
\item \textbf{Depth adversary} $256 \rightarrow 64 \rightarrow 1$ behind the same GradReverse, predicting cell-cycle-adjusted $\log(\text{counts})$.
|
| 99 |
+
\end{itemize}
|
| 100 |
+
|
| 101 |
+
Two variants share this architecture and differ only in the input to the trunk: $\panda_\text{PCA}$ uses the 50-d PCA vector alone; $\panda_\text{Marker}$ concatenates the PCA vector with a curated per-cell marker-score vector $\mathbf{m}$ (canonical marker gene modules for the tissue, scored via \texttt{sc.tl.score\_genes}).
|
| 102 |
+
|
| 103 |
+
\subsection{Losses and training curriculum}
|
| 104 |
+
|
| 105 |
+
The composite loss is
|
| 106 |
+
\[
|
| 107 |
+
\mathcal{L} \;=\; \mathcal{L}_{\text{SupCon}} + \lambda_{\text{V}}\mathcal{L}_{\text{VICReg}} + \lambda_{\text{CE}}\mathcal{L}_{\text{CE}} + \lambda_{\text{P}}\mathcal{L}_{\text{proto}} + \lambda_{\text{D}}\mathcal{L}_{\text{dom}} + \lambda_{\text{Z}}\mathcal{L}_{\text{depth}} + \lambda_{\text{H}}\mathcal{L}_{\text{HSIC}},
|
| 108 |
+
\]
|
| 109 |
+
with $(\lambda_{\text{V}}, \lambda_{\text{CE}}, \lambda_{\text{P}}, \lambda_{\text{D}}, \lambda_{\text{Z}}, \lambda_{\text{H}}) = (1.0, 0.4, 0.6, 1.0, 0.3, 0.05)$. Each term has a distinct role:
|
| 110 |
+
\begin{enumerate}
|
| 111 |
+
\item $\mathcal{L}_{\text{SupCon}}$ --- class-balanced supervised contrastive loss on 128-d projections:
|
| 112 |
+
\[
|
| 113 |
+
-\sum_i \tfrac{1}{|P(i)|}\sum_{p \in P(i)} \log \tfrac{\exp(z_i\!\cdot z_p / \tau)}{\sum_{a \neq i} \exp(z_i\!\cdot z_a / \tau)};
|
| 114 |
+
\]
|
| 115 |
+
positives $P(i)$ drawn from $\geq 2$ datasets per class per batch, per-class weight $w_c = 1/\sqrt{n_c}$, $\tau = 0.1$.
|
| 116 |
+
\item $\mathcal{L}_{\text{VICReg}}$ --- variance-invariance-covariance regularization on the 128-d projections; keeps each dimension informative and decorrelated.
|
| 117 |
+
\item $\mathcal{L}_{\text{CE}}$ --- classifier cross-entropy on the auxiliary head; weight blended $0.5\cdot 1/\sqrt{n_c} + 0.5$ to balance rare classes.
|
| 118 |
+
\item $\mathcal{L}_{\text{proto}}$ --- prototype-InfoNCE that pulls each projection toward its class-prototype: $-\log \tfrac{\exp(z_i\!\cdot\tilde p_{y_i}/\tau_p)}{\sum_c \exp(z_i\!\cdot\tilde p_c/\tau_p)}$ with EMA momentum $0.99$, $\tau_p = 0.07$.
|
| 119 |
+
\item $\mathcal{L}_{\text{dom}}$ --- cross-entropy against the dataset adversary through a gradient-reversal layer (GRL); trunk sees $-\lambda_{\text{GRL}}\nabla$ so it removes dataset-identifiable structure.
|
| 120 |
+
\item $\mathcal{L}_{\text{depth}}$ --- MSE against the depth adversary predicting cell-cycle-adjusted $\log(\text{counts})$, also through GRL.
|
| 121 |
+
\item $\mathcal{L}_{\text{HSIC}}$ --- biased Hilbert-Schmidt Independence Criterion (HSIC) between representation and $\log(\text{counts})$; drives depth-representation dependence to zero as a stronger complement to the depth adversary.
|
| 122 |
+
\end{enumerate}
|
| 123 |
+
|
| 124 |
+
Training curriculum, gated by go/no-go metrics: (i) SupCon $+$ VICReg $+$ CE warmup until leave-one-dataset-out (LODO) macro-AUROC $\geq 0.80$; (ii) engage $\mathcal{L}_{\text{proto}}$ with EMA prototypes once RankMe $\geq 60$; (iii) ramp $\lambda_{\text{GRL}}$ from 0 to 1 to engage the adversarial and HSIC terms with projection-space mixup ($\alpha = 0.1$) and depth-jitter augmentation; (iv) low-lr fine-tune with frozen prototypes. Sampler: hybrid $P\times K\times D$ batch sampler with $\geq 6$ guaranteed cells per class per batch plus 96 natural-frequency slots.
|
| 125 |
+
|
| 126 |
+
\subsection{Inference}
|
| 127 |
+
|
| 128 |
+
For a target dataset:
|
| 129 |
+
\begin{enumerate}
|
| 130 |
+
\item Project raw counts to the frozen corpus shared HVG index (zero-impute missing genes); compute per-cell \texttt{missing\_hvg\_frac}.
|
| 131 |
+
\item Log-normalize with corpus per-gene mean/std; clip to $[-10, 10]$.
|
| 132 |
+
\item Frozen sample-fit PCA transform to 50-d (concatenate marker channel for the Marker variant).
|
| 133 |
+
\item Forward through frozen \panda{} to obtain 128-d projections $z$.
|
| 134 |
+
\item Compute cosine similarity to $K$ frozen prototypes: $\cos_{ic} = z_i \cdot \tilde{p}_c$.
|
| 135 |
+
\item Temperature-scale ($\tau = 0.07$) and softmax to obtain class probabilities.
|
| 136 |
+
\item Apply black-box shift estimation (BBSE) label-shift correction: $q_D(y) = W^{-1} \hat{p}_D(\hat{y})$.
|
| 137 |
+
\item Confidence-gate at $\cos < 0.3$; abstained cells enter downstream novel-population discovery.
|
| 138 |
+
\end{enumerate}
|
| 139 |
+
|
| 140 |
+
\section{Corpora (100\% paper-labeled)}
|
| 141 |
+
\label{sec:corpus}
|
| 142 |
+
|
| 143 |
+
Every training cell carries a label taken from its source paper's own supplementary tables, deposited annotations, or public cell-metadata release. Datasets without paper-provided labels are excluded rather than filled in by our marker scorer. This eliminates one full class of label noise (heuristic marker-score misassignment) and turns the corpus into a benchmark on which zero-shot F1 is directly interpretable.
|
| 144 |
+
|
| 145 |
+
Per tissue: (a) download published scRNA-seq via GEO/ArrayExpress; (b) per-dataset load with format-specific loaders $+$ QC (min\_genes $=200$, min\_cells $=3$, mt\% $<20\%$); (c) obtain paper labels from supplementary tables and harmonise to the corpus vocabulary; (d) compute shared highly-variable genes (HVGs) via a union$+$majority rule where any gene ranking in the top-4000 of $\geq n/2$ datasets is eligible, then forced-include a curated list of 40--50 canonical markers per system; (e) sample-fit PCA on a 30--50k cell subsample; (f) transform all datasets; (g) hold out truly labeled external datasets (Baron test-half, Nestorowa, Sulic, Belote, Veres held-out slice) for zero-shot evaluation.
|
| 146 |
+
|
| 147 |
+
\textbf{Pan-skin (45{,}387 cells, 13 classes)}: Sulic GSE212673 (E14.5 dorsal skin), Merkel GSE201447, MCA GSE108097 neonatal skin \cite{han2018mca}, Joost GSE67602 \cite{joost2016}, Haensel/Annusver GSE142471 \cite{haensel2020skin}, and Belote GSE151091 \cite{belote2021} as the melanocyte anchor. Classes: basal-IFE, spinous, granular, HF-placode, HF-ORS, endothelial, immune, sebaceous, fibroblast-papillary, fibroblast-reticular, melanocyte, melanoblast, melanocyte-precursor.
|
| 148 |
+
|
| 149 |
+
\textbf{Pan-hematopoietic (192{,}833 cells, 15 classes)}: Weinreb LARRY GSE140802, Baccin GSE122465 \cite{baccin2020}, and Tabula Muris Senis bone marrow GSE132042 \cite{tms2020}. Classes: long-term hematopoietic stem cell (LT-HSC), multipotent progenitor (MPP), myeloid, monocyte, macrophage, basophil-mast, erythroid, megakaryocyte, T-cell, naive-B, pro-B, lymphoid, fibroblast, stromal, endothelial.
|
| 150 |
+
|
| 151 |
+
\textbf{Pan-pancreatic (120{,}611 cells, 20 classes)}: Baron 2016 mouse-train half GSE84133, Bastidas GSE132188 E15.5 \cite{bastidas2019}, Byrnes 2018 GSE101099 paper-labeled subset \cite{byrnes2018}, Yu 2021 GSE139627 paper-labeled subset \cite{yu2021}, MIA GSE211796 paper-labeled subset \cite{hrovatin2023mia}, and Veres 2019 GSE114412 --- of which 57{,}297 cells joined the training corpus and \textbf{12{,}297 labeled cells were held out} as an external zero-shot test target. Classes: alpha, beta, delta, gamma, epsilon, ductal, acinar, endothelial, immune, mesenchyme, alpha\_progenitor, beta\_progenitor, adult-alpha, adult-beta, endocrine-progenitor, endocrine-progenitor-primed, pancreatic-progenitor, proliferating, Fev-EP, exocrine.
|
| 152 |
+
|
| 153 |
+
\section{Held-out 5-fold cross-validation}
|
| 154 |
+
|
| 155 |
+
\subsection{Pan-skin: 13 classes, 45{,}387 cells}
|
| 156 |
+
|
| 157 |
+
Stratified 5-fold cross-validation with \panda{} retrained from scratch on 80\% train per fold (5 epochs per fold), evaluated on the held-out 20\%:
|
| 158 |
+
\begin{center}
|
| 159 |
+
\begin{tabular}{lccc}
|
| 160 |
+
\toprule
|
| 161 |
+
Variant & Accuracy & Macro F1 & Macro AUROC \\
|
| 162 |
+
\midrule
|
| 163 |
+
\panda{}-PCA & $0.9453 \pm 0.0020$ & $0.9150$ & $0.9975$ \\
|
| 164 |
+
\panda{}-Marker & $\mathbf{0.9476 \pm 0.0027}$ & $0.9139$ & $\mathbf{0.9977}$ \\
|
| 165 |
+
\bottomrule
|
| 166 |
+
\end{tabular}
|
| 167 |
+
\end{center}
|
| 168 |
+
Results at \texttt{discovery/pan\_skin/\{pca,marker\}/cv\_5fold.json}. Per-class F1 across all three systems in Figure~\ref{fig:cv}.
|
| 169 |
+
|
| 170 |
+
\begin{figure}[!htb]
|
| 171 |
+
\centering
|
| 172 |
+
\includegraphics[width=\textwidth]{figures/fig1_confusion_matrices.pdf}
|
| 173 |
+
\caption[Per-class F1 across three systems]{Per-class F1 on held-out 5-fold CV for the three multi-dataset systems.\\
|
| 174 |
+
\textbf{Skin} (13 classes, mean acc $=0.948$): F1 $\geq 0.88$ on 10/13 classes; granular, melanocyte-precursor, and sebaceous fall below on low support.\\
|
| 175 |
+
\textbf{Pan-hematopoietic} (15 classes, mean acc $=0.938$): F1 $\geq 0.88$ on the major lineages (MPP, LT-HSC, erythroid, myeloid, monocyte, basophil-mast, pro-B, T-cell, naive-B, endothelial); lymphoid, macrophage, and megakaryocyte sit just below.\\
|
| 176 |
+
\textbf{Pan-pancreatic} (20 classes, mean acc $=0.794$): the hardest system --- only the largest identities (adult-beta, alpha\_progenitor, adult-alpha, pancreatic-progenitor, beta\_progenitor, endothelial, exocrine) exceed F1 $=0.88$; fine-grained endocrine sub-classes trade support for granularity.}
|
| 177 |
+
\label{fig:cv}
|
| 178 |
+
\end{figure}
|
| 179 |
+
|
| 180 |
+
\subsection{Pan-hematopoietic: 15 classes, 192{,}833 cells}
|
| 181 |
+
|
| 182 |
+
5-fold stratified CV on the 15-class Weinreb + Baccin + TMS corpus:
|
| 183 |
+
\begin{center}
|
| 184 |
+
\begin{tabular}{lccc}
|
| 185 |
+
\toprule
|
| 186 |
+
Variant & Accuracy & Macro F1 & Macro AUROC \\
|
| 187 |
+
\midrule
|
| 188 |
+
\panda{}-PCA & $0.9384 \pm 0.0010$ & $0.9305$ & $0.9967$ \\
|
| 189 |
+
\panda{}-Marker & $\mathbf{0.9376 \pm 0.0024}$ & $\mathbf{0.9317}$ & $\mathbf{0.9969}$ \\
|
| 190 |
+
\bottomrule
|
| 191 |
+
\end{tabular}
|
| 192 |
+
\end{center}
|
| 193 |
+
PCA and Marker are within one standard deviation on accuracy; Marker wins on macro F1 and AUROC. Results at \texttt{discovery/hematopoiesis/\{pca,marker\}/cv\_5fold.json}.
|
| 194 |
+
|
| 195 |
+
\subsection{Pan-pancreatic: 20 classes, 120{,}611 cells}
|
| 196 |
+
|
| 197 |
+
57{,}297 Veres cells join the training corpus alongside Baron/Bastidas/Byrnes/Yu/MIA; 12{,}297 labeled Veres cells are held out for \S\ref{sec:zeroshot-veres}. 5-fold stratified CV:
|
| 198 |
+
\begin{center}
|
| 199 |
+
\begin{tabular}{lccc}
|
| 200 |
+
\toprule
|
| 201 |
+
Variant & Accuracy & Macro F1 & Macro AUROC \\
|
| 202 |
+
\midrule
|
| 203 |
+
\panda{}-PCA & $0.7522 \pm 0.0112$ & $0.7362$ & $0.9688$ \\
|
| 204 |
+
\panda{}-Marker & $\mathbf{0.7942 \pm 0.0025}$ & $\mathbf{0.7645}$ & $\mathbf{0.9741}$ \\
|
| 205 |
+
\bottomrule
|
| 206 |
+
\end{tabular}
|
| 207 |
+
\end{center}
|
| 208 |
+
Marker beats PCA by $+4.2\%$ accuracy and $+2.8\%$ F1 --- the largest marker gain of any system, consistent with pancreatic endocrine sub-lineages differing on low-variance TFs (Nkx6-1, Mnx1, Arx) that benefit most from the direct marker channel.
|
| 209 |
+
|
| 210 |
+
\subsection{Multi-seed rigor}
|
| 211 |
+
\label{sec:multiseed}
|
| 212 |
+
|
| 213 |
+
We repeat every system's 5-fold CV with independent seeds affecting fold assignment (\texttt{StratifiedKFold} \texttt{random\_state}) and model initialisation. Across 35 fold-comparisons (7 seed-runs $\times$ 5 folds; 2 seeds each for skin and pancreas, 3 for HSC), Marker beats PCA on accuracy in \textbf{33/35 folds}, unevenly distributed: pancreas Marker wins every fold across every seed ($+0.04$ to $+0.06$ per-seed accuracy); HSC is a tie at 5 epochs (Marker within $\pm 0.001$ of PCA); skin Marker leads on accuracy while PCA slightly leads on macro-F1 in some seeds. The marker channel is a mostly-positive, system-dependent choice, with the largest lift on pancreatic endocrine subtypes. {\sloppy Per-configuration artefacts at \pathsplit{discovery/\{system\}/\{variant\}/cv\_5fold\{,\_seed1,\_seed2\}.json}.\par}
|
| 214 |
+
|
| 215 |
+
\section{Held-out labeled targets}
|
| 216 |
+
|
| 217 |
+
A single pass of the pan-corpus \panda{} checkpoint on labeled datasets held out from the start: no fold-training, no per-target ensembling. Baron, Sulic, Belote, and Nestorowa are strict zero-shot (source contributed zero training cells); Veres (\S\ref{sec:zeroshot-veres}) is a held-out slice (57{,}297 cells trained, 12{,}297 held out).
|
| 218 |
+
|
| 219 |
+
\subsection{Baron test-half (pancreas)}
|
| 220 |
+
\label{sec:zeroshot-baron}
|
| 221 |
+
|
| 222 |
+
The 943-cell mouse test half of Baron 2016 GSE84133 was held out of the pancreatic corpus from the initial build. We score against the paper's canonical \texttt{assigned\_cluster} labels after harmonising PANDA's adult sub-prototypes (\texttt{adult-beta}$\to$\texttt{beta}, \texttt{adult-alpha}$\to$\texttt{alpha}) into Baron's flat vocabulary. Of the 943 cells, \textbf{906 are evaluated on the shared-class vocabulary} (37 cells whose Baron labels do not exist in the pan-pancreatic class list are excluded from accuracy/F1 to avoid off-vocabulary penalties).
|
| 223 |
+
|
| 224 |
+
\begin{center}
|
| 225 |
+
\begin{tabular}{lcc}
|
| 226 |
+
\toprule
|
| 227 |
+
Variant & Accuracy & Macro F1 \\
|
| 228 |
+
\midrule
|
| 229 |
+
\panda{}-PCA & $\mathbf{0.9426}$ & $\mathbf{0.6027}$ \\
|
| 230 |
+
\panda{}-Marker & $0.8190$ & $0.5888$ \\
|
| 231 |
+
\bottomrule
|
| 232 |
+
\end{tabular}
|
| 233 |
+
\end{center}
|
| 234 |
+
|
| 235 |
+
Baron is the one zero-shot target where PCA leads on both accuracy and macro-F1 ($+12.4$ acc, $+1.4$ F1). Marker's sharper adult-vs-embryonic prototype geometry splits rare Baron classes across two nearby prototypes (adult-$\alpha$/juvenile-$\alpha$; adult-$\beta$/juvenile-$\beta$); on a 906-cell slice with imbalanced supports this geometry costs both accuracy and F1. On the larger Veres held-out (\S\ref{sec:zeroshot-veres}) the same sub-prototype separation is discriminative and Marker wins.
|
| 236 |
+
|
| 237 |
+
\subsection{Veres held-out (pancreas)}
|
| 238 |
+
\label{sec:zeroshot-veres}
|
| 239 |
+
|
| 240 |
+
12{,}297 Veres 2019 GSE114412 cells were held out with paper labels preserved. This is a held-out-slice benchmark (57{,}297 Veres cells trained), not a strict zero-shot like Baron/Sulic/Belote/Nestorowa; it is nonetheless the largest labeled held-out pancreatic benchmark in the paper.
|
| 241 |
+
|
| 242 |
+
\begin{center}
|
| 243 |
+
\begin{tabular}{lcc}
|
| 244 |
+
\toprule
|
| 245 |
+
Variant & Accuracy & Macro F1 \\
|
| 246 |
+
\midrule
|
| 247 |
+
\panda{}-PCA & $0.9065$ & $0.8922$ \\
|
| 248 |
+
\panda{}-Marker & $\mathbf{0.9131}$ & $\mathbf{0.9001}$ \\
|
| 249 |
+
\bottomrule
|
| 250 |
+
\end{tabular}
|
| 251 |
+
\end{center}
|
| 252 |
+
|
| 253 |
+
At 12{,}297-cell scale with 20 canonical Veres classes, Marker beats PCA by $+0.66\%$ accuracy and $+0.79\%$ F1. Macro F1 $= 0.900$ is the strongest labeled held-out pancreatic F1 in this paper. Results at \texttt{discovery/pancreas/\{pca,marker\}/veres\_summary.json}.
|
| 254 |
+
|
| 255 |
+
\subsection{Nestorowa Smart-seq2 (hematopoiesis)}
|
| 256 |
+
\label{sec:zeroshot-nestorowa}
|
| 257 |
+
|
| 258 |
+
Nestorowa 2016 GSE81682, 1{,}170 Smart-seq2 mouse HSPCs with FACS labels (LT-HSC vs.\ HSPC gate). The corpus contains an LT-HSC class populated from Tabula Muris Senis bone-marrow paper labels; no per-target anchor is added.
|
| 259 |
+
|
| 260 |
+
\begin{center}
|
| 261 |
+
\begin{tabular}{lc}
|
| 262 |
+
\toprule
|
| 263 |
+
Variant & LT-HSC recall (66 FACS-labeled LT-HSC cells) \\
|
| 264 |
+
\midrule
|
| 265 |
+
\panda{}-PCA & $0.045$ \\
|
| 266 |
+
\panda{}-Marker & $\mathbf{0.212}$ \\
|
| 267 |
+
\bottomrule
|
| 268 |
+
\end{tabular}
|
| 269 |
+
\end{center}
|
| 270 |
+
|
| 271 |
+
Both variants struggle: the Smart-seq2 LT-HSC FACS gate is transcriptomically broader than the TMS 10x LT-HSC prototype learned from the corpus. The 0.212 figure is per-class recall on 66 FACS-labeled LT-HSC cells (14/66 correct). Marker's $\sim\!5\times$ improvement over PCA (14/66 vs 3/66; Wilson 95\% CIs $[0.12, 0.32]$ vs $[0.01, 0.13]$) suggests the marker channel partially bridges the cross-platform mismatch by reading canonical LT-HSC genes (Hlf, Mecom, Mpl) that PCA compresses into its 50-component bottleneck. A Smart-seq2 LT-HSC training anchor is the natural next step.
|
| 272 |
+
|
| 273 |
+
\subsection{Belote melanocyte (skin)}
|
| 274 |
+
\label{sec:zeroshot-belote}
|
| 275 |
+
|
| 276 |
+
Belote 2021 GSE151091 \cite{belote2021}: 6{,}088 human melanocyte-lineage cells across mel / melanoblast / melanocyte-precursor sub-classes, evaluated on cells never seen in training.
|
| 277 |
+
|
| 278 |
+
\begin{center}
|
| 279 |
+
\begin{tabular}{lcc}
|
| 280 |
+
\toprule
|
| 281 |
+
Variant & Accuracy & Macro F1 \\
|
| 282 |
+
\midrule
|
| 283 |
+
\panda{}-PCA & $\mathbf{0.969}$ & $0.395$ \\
|
| 284 |
+
\panda{}-Marker & $0.955$ & $0.340$ \\
|
| 285 |
+
\bottomrule
|
| 286 |
+
\end{tabular}
|
| 287 |
+
\end{center}
|
| 288 |
+
|
| 289 |
+
High accuracy reflects correct assignment to the dominant \emph{melanocyte} class; low macro F1 is a class-imbalance artefact across the three melanocyte sub-classes.
|
| 290 |
+
|
| 291 |
+
\subsection{Sulic (skin)}
|
| 292 |
+
\label{sec:zeroshot-sulic}
|
| 293 |
+
|
| 294 |
+
Sulic 2023 GSE212673, 4{,}683 E14.5 mouse dorsal skin cells; 500 anchor cells serve as an HF-placode anchor in training, and 4{,}183 cells are held out for zero-shot evaluation.
|
| 295 |
+
|
| 296 |
+
\begin{center}
|
| 297 |
+
\begin{tabular}{lcc}
|
| 298 |
+
\toprule
|
| 299 |
+
Variant & Accuracy & Macro F1 \\
|
| 300 |
+
\midrule
|
| 301 |
+
\panda{}-PCA & $\mathbf{0.940}$ & $\mathbf{0.930}$ \\
|
| 302 |
+
\panda{}-Marker & $0.895$ & $0.871$ \\
|
| 303 |
+
\bottomrule
|
| 304 |
+
\end{tabular}
|
| 305 |
+
\end{center}
|
| 306 |
+
|
| 307 |
+
Zero-shot macro F1 $= 0.930$ on 4{,}183 held-out cells is the strongest labeled zero-shot skin F1 in this paper.
|
| 308 |
+
|
| 309 |
+
\section{Discovery target: Dingwall En1-cKO}
|
| 310 |
+
\label{sec:dingwall}
|
| 311 |
+
|
| 312 |
+
\subsection{Zero-shot classification}
|
| 313 |
+
\label{sec:dingwall-zshot-class}
|
| 314 |
+
|
| 315 |
+
Held-out target: Dingwall \emph{et al.}\ 2024 \cite{dingwall2024en1cko}, \emph{Developmental Cell} 59(1):20--32.e6 (Kamberov lab), GSE220977: 25{,}800 cells P2.5 volar hindpaw snRNA-seq after per-sample QC. Genotype mapping: WT $=$ \{GSM6833478, 79, 80, 81\} ($n=15{,}400$; 5{,}100 $+$ 3{,}500 $+$ 3{,}300 $+$ 3{,}500), En1-cKO $=$ \{GSM6833482, 83\} ($n=10{,}400$; 5{,}400 $+$ 5{,}000). Baseline cKO fraction on the full 25{,}800-cell target is $0.403$; on the Seurat-replica dermal subset (14{,}252 cells) it is $0.382$. The Dingwall data never touches training, HVG selection, or PCA fit.
|
| 316 |
+
|
| 317 |
+
Predicted class distribution (25{,}800 cells; the pan-skin vocabulary $\{$HF-ORS, HF-placode, basal-IFE, endothelial, fibroblast-papillary, fibroblast-reticular, granular, immune, melanoblast, melanocyte, melanocyte-precursor, sebaceous, spinous$\}$; fibroblast-papillary is in the vocabulary but has zero Dingwall predictions):
|
| 318 |
+
\begin{center}
|
| 319 |
+
\begin{tabular}{lrr}
|
| 320 |
+
\toprule
|
| 321 |
+
Class & $n$ & Fraction \\
|
| 322 |
+
\midrule
|
| 323 |
+
fibroblast-reticular & 13{,}798 & 53.5\% \\
|
| 324 |
+
melanoblast & 5{,}143 & 19.9\% \\
|
| 325 |
+
endothelial & 3{,}608 & 14.0\% \\
|
| 326 |
+
immune & 1{,}039 & 4.0\% \\
|
| 327 |
+
HF-placode & 546 & 2.1\% \\
|
| 328 |
+
spinous & 342 & 1.3\% \\
|
| 329 |
+
basal-IFE & 335 & 1.3\% \\
|
| 330 |
+
HF-ORS & 328 & 1.3\% \\
|
| 331 |
+
granular & 262 & 1.0\% \\
|
| 332 |
+
melanocyte & 190 & 0.7\% \\
|
| 333 |
+
melanocyte-precursor & 123 & 0.5\% \\
|
| 334 |
+
sebaceous & 86 & 0.3\% \\
|
| 335 |
+
\bottomrule
|
| 336 |
+
\end{tabular}
|
| 337 |
+
\end{center}
|
| 338 |
+
|
| 339 |
+
\begin{figure}[!htb]
|
| 340 |
+
\centering
|
| 341 |
+
\includegraphics[width=\textwidth]{figures/fig5_dingwall_umap.pdf}
|
| 342 |
+
\caption[UMAP of Dingwall projection]{UMAP of \panda{}'s 128-d projection of Dingwall ($n=25{,}800$ cells).\\
|
| 343 |
+
\textbf{Left:} cells coloured by BBSE-corrected predicted class.\\
|
| 344 |
+
\textbf{Right:} cells coloured by En1 genotype (WT vs.\ En1-cKO).\\
|
| 345 |
+
\textbf{Observation:} clusters correspond to distinct populations at expected proportions; WT and cKO cells share cluster occupancy but differ in local density (melanocyte, HF-placode, spinous).}
|
| 346 |
+
\label{fig:umap}
|
| 347 |
+
\end{figure}
|
| 348 |
+
|
| 349 |
+
Wilcoxon DE per predicted class on raw Dingwall expression recovers canonical marker sets without those genes being supplied to the model: fibroblast-reticular (5/6 canonical: Dcn, Fbn1, Postn, Lum, Col1a2); endothelial (5/7: Kdr, Tie1, Cdh5, Tek, Pecam1); basal-IFE (3/6: Krt14, Trp63, Krt5); immune (Ptprc, Adgre1 $+$ Mrc1, F13a1 macrophage).
|
| 350 |
+
|
| 351 |
+
\subsection{Class-level enrichment: En1-cKO vs.\ WT}
|
| 352 |
+
\label{sec:dingwall-fisher}
|
| 353 |
+
|
| 354 |
+
Fisher exact tests of PANDA class calls vs.\ the $\sim 40:60$ cKO:WT background; $\log_2$ fold-change is the class cKO/WT odds relative to background odds; $p_\text{adj}$ is Benjamini--Hochberg over 12 classes:
|
| 355 |
+
\begin{center}
|
| 356 |
+
\small
|
| 357 |
+
\begin{tabular}{lrrrl}
|
| 358 |
+
\toprule
|
| 359 |
+
Class & $n$ & $\log_2$ fold cKO/WT & Fisher $p_\text{adj}$ & Direction \\
|
| 360 |
+
\midrule
|
| 361 |
+
spinous & 342 & $-0.87$ & $\mathbf{5.1 \times 10^{-6}}$ & $\downarrow$ cKO \\
|
| 362 |
+
melanocyte & 190 & $+0.79$ & $\mathbf{1.2 \times 10^{-3}}$ & $\uparrow$ cKO ($\sim 1.7\times$) \\
|
| 363 |
+
HF-placode & 546 & $+0.43$ & $\mathbf{2.7 \times 10^{-3}}$ & $\uparrow$ cKO ($\sim 1.3\times$) \\
|
| 364 |
+
fibroblast-reticular & 13{,}798 & $-0.09$ & $\mathbf{3.9 \times 10^{-2}}$ & $\downarrow$ cKO \\
|
| 365 |
+
granular & 262 & $-0.40$ & $0.088$ & (ns) \\
|
| 366 |
+
basal-IFE & 335 & $+0.30$ & $0.128$ & (ns) \\
|
| 367 |
+
endothelial & 3{,}608 & $+0.09$ & $0.152$ & (ns) \\
|
| 368 |
+
HF-ORS & 328 & $-0.27$ & $0.169$ & (ns) \\
|
| 369 |
+
melanoblast & 5{,}143 & $+0.05$ & $0.328$ & (ns) \\
|
| 370 |
+
immune & 1{,}039 & $+0.10$ & $0.328$ & (ns) \\
|
| 371 |
+
melanocyte-precursor & 123 & $-0.23$ & $0.446$ & (ns) \\
|
| 372 |
+
sebaceous & 86 & $-0.12$ & $0.742$ & (ns) \\
|
| 373 |
+
\bottomrule
|
| 374 |
+
\end{tabular}
|
| 375 |
+
\end{center}
|
| 376 |
+
|
| 377 |
+
Spinous keratinocyte depletion is the strongest compositional signal ($\log_2\!\text{fc} = -0.87$, $p_\text{adj} = 5.1 \times 10^{-6}$), consistent with En1 loss impairing spinous-layer terminal differentiation. Melanocyte and HF-placode compartments are modestly enriched in cKO ($1.7\times$ and $1.3\times$ respectively), and reticular fibroblast is modestly depleted.
|
| 378 |
+
|
| 379 |
+
\subsection{Multi-class pathway analysis}
|
| 380 |
+
\label{sec:dingwall-pathway}
|
| 381 |
+
|
| 382 |
+
\texttt{sc.tl.score\_genes} on an expanded 15$+$ canonical pathway module set, Mann--Whitney U tested En1-cKO vs.\ WT within each PANDA-predicted class. Top hits at $p_\text{adj} < 0.01$:
|
| 383 |
+
\begin{center}
|
| 384 |
+
\small
|
| 385 |
+
\setlength{\tabcolsep}{4pt}
|
| 386 |
+
\begin{tabular}{p{2.9cm}p{3.1cm}p{1.1cm}p{2.3cm}p{4.7cm}}
|
| 387 |
+
\toprule
|
| 388 |
+
Class & Pathway & $\Delta$ & MannU $p_\text{adj}$ & Interpretation \\
|
| 389 |
+
\midrule
|
| 390 |
+
HF-placode & Eda\_ectodysplasin & $+0.142$ & $\mathbf{3.0 \times 10^{-3}}$ & Ectopic placode-signal reactivation \\
|
| 391 |
+
fibroblast-reticular & Eda\_ectodysplasin & $+0.029$ & $\mathbf{4.0 \times 10^{-6}}$ & Dermis reactivates placode signal \\
|
| 392 |
+
melanoblast & Sweat\_gland & $+0.022$ & $\mathbf{3.7 \times 10^{-6}}$ & Ectopic eccrine program in melanoblast lineage \\
|
| 393 |
+
melanoblast & Eda\_ectodysplasin & $+0.044$ & $\mathbf{9.1 \times 10^{-4}}$ & Ectopic Eda in melanoblast \\
|
| 394 |
+
melanoblast & BMP\_signaling & $+0.028$ & $\mathbf{1.1 \times 10^{-4}}$ & BMP derepression in melanoblast \\
|
| 395 |
+
fibroblast-reticular & Basal\_keratinocyte & $+0.034$ & $\mathbf{6.2 \times 10^{-6}}$ & Dermal cells acquire basal-keratinocyte signal \\
|
| 396 |
+
\bottomrule
|
| 397 |
+
\end{tabular}
|
| 398 |
+
\end{center}
|
| 399 |
+
|
| 400 |
+
\textbf{Interpretation:} Eda\_ectodysplasin derepression is widespread across HF-placode, fibroblast-reticular, and melanoblast --- consistent with En1's role as a \emph{spatial repressor of the ectodermal-appendage program}. Sweat\_gland derepression appears in the melanoblast lineage (novel). {\sloppy All module $\times$ class contrasts at \pathsplit{discovery/pan\_skin/marker/57\_pathway\_class\_by\_module\_padj.tsv} and \pathsplit{57\_pathway\_class\_by\_module\_delta.tsv}.\par}
|
| 401 |
+
|
| 402 |
+
\subsection{Dingwall EDEN validation: three complementary lines of evidence}
|
| 403 |
+
\label{sec:dingwall-eden}
|
| 404 |
+
|
| 405 |
+
Dingwall's paper identifies an ``embryonic dermal En1$^+$ niche'' (EDEN, Derm10 in the paper's clustering) that is dramatically cKO-depleted and contains the dermal signalling activity supporting sweat-gland placode induction. Because our pan-skin corpus does not include an EDEN class prototype directly, we validate the EDEN phenotype through two positive lines of evidence (B, C) and one negative scope check (A) that motivates the other two.
|
| 406 |
+
|
| 407 |
+
\textbf{Line A --- Scope check: post-hoc scoring of the pan-tissue fibroblast prototype does \emph{not} recover EDEN.} PANDA's pan-skin fibroblast predictions on Dingwall (13{,}798 cells) were scored on Dingwall's minimal Data-S1C EDEN module (S100a4 $+$ Tnc $+$ Pdgfra). The top-5\% (690 cells) show no significant WT enrichment (WT fraction $= 0.612$ vs.\ dermal-fibroblast baseline $0.604$). This null result establishes that the pan-tissue fibroblast prototype is calibrated too broadly to resolve EDEN by itself and motivates Lines B and C rather than validating the phenotype. Artefact: \texttt{discovery/pan\_skin/marker/98\_eden\_summary.json}.
|
| 408 |
+
|
| 409 |
+
\textbf{Line B --- Independent scanpy reproduction of Dingwall's clustering.} We reproduced Dingwall's Seurat pipeline in scanpy (LogNormalize $\to$ HVG $=2000$ $\to$ PCA $=40$ $\to$ Harmony per-sample $\to$ Leiden res $=0.7 \to$ 23 top-level clusters; dermal subset re-Leidened into 12 Derm subclusters mapped to Dingwall's Data-S1C Derm0--Derm11 panels by top-50-marker Jaccard). Derm10 recovers at \textbf{206 WT / 30 cKO dermal cells} $=$ $\mathbf{4.32\times}$ WT enrichment, Fisher $p = 2.25\!\times\!10^{-18}$; direction matches Dingwall's paper (their effect is on the 45{,}370-cell superset, ours on the 25{,}800-cell QC subset). Dermal baseline in our replica: $8{,}806$ WT / $5{,}446$ cKO. Artefact: \texttt{data/processed/dingwall\_replica/replica\_cluster\_20\_qc.json}.
|
| 410 |
+
|
| 411 |
+
\textbf{Line C --- PANDA trained directly on the reproduction's Derm labels.} \panda{}-Marker trained on the reproduction's Derm0--Derm11 labels with a 70/30 genotype-stratified split ($9{,}965$ train / $4{,}287$ test) reaches $\mathbf{81.4\%}$ test accuracy across 10 Derm classes. On held-out cells, predicted Derm10 shows $\mathbf{5.20\times}$ WT-enrichment odds-ratio (Fisher $p = 5.8\!\times\!10^{-9}$); scored against true reproduction labels on the same cells, Derm10 shows $4.34\times$ WT enrichment ($p = 2.9\!\times\!10^{-6}$). The EDEN depletion phenotype is reproduced at inference time from cells the base pan-skin corpus never saw as EDEN. Artefact: \texttt{discovery/pan\_skin/marker/104\_dingwall\_derm\_summary.json}.
|
| 412 |
+
|
| 413 |
+
Lines B and C together show that Derm10's En1-cKO depletion is a real, learnable, reproducible property; Line A shows the base pan-tissue fibroblast prototype is too broad to resolve it on its own.
|
| 414 |
+
|
| 415 |
+
\subsection{Primary EDEN candidate: Derm2}
|
| 416 |
+
\label{sec:dingwall-primary-eden}
|
| 417 |
+
|
| 418 |
+
We next asked whether \panda{} could extend Dingwall's En1-dependence phenotype up the paper's own developmental lineage. Dingwall's Fig 5 Slingshot pseudotime places Derm10 (Secondary EDEN, cKO-depleted, statistically tested) downstream of a lineage running Derm3 $\to$ Derm6 $\to$ Derm9 $\to$ Derm2 $\to$ Derm10, but the upstream clusters were not statistically tested for cKO depletion in the paper.
|
| 419 |
+
|
| 420 |
+
Sub-clustering the reproduction's 14{,}252 dermal cells at Leiden resolution $=1.5$ and scoring each of the 12 subclusters against Dingwall's Data-S1C Derm panels (top-30 markers per panel) identifies a subcluster whose top Wilcoxon markers match \textbf{Derm2 at 7/10 top-marker overlap} (Enpp2, Bmp5, Hpse2, Cacna2d3, Sned1, Rspo2, Slc24a3). This Derm2-matched subcluster contains \textbf{1{,}586 cells (1{,}051 WT / 535 cKO)}, a cKO fraction of $0.337$ vs.\ the dermal baseline $0.382$ --- \textbf{$1.25\times$ WT enrichment, Fisher $p = 9.84 \times 10^{-5}$}.
|
| 421 |
+
|
| 422 |
+
\begin{center}
|
| 423 |
+
\small
|
| 424 |
+
\setlength{\tabcolsep}{4pt}
|
| 425 |
+
\begin{tabular}{p{2.0cm}p{3.0cm}p{4.9cm}p{2.3cm}p{2.3cm}}
|
| 426 |
+
\toprule
|
| 427 |
+
Leiden sub-cluster & Dingwall panel & Marker overlap & Fisher $p$ & Direction \\
|
| 428 |
+
\midrule
|
| 429 |
+
Derm2 match & \textbf{Derm2} (Primary EDEN candidate) & \textbf{7/10} (Enpp2, Bmp5, Hpse2, Cacna2d3, Sned1, Rspo2, Slc24a3) & $\mathbf{9.84 \times 10^{-5}}$ & $\downarrow$ cKO ($1.25\times$) \\
|
| 430 |
+
\bottomrule
|
| 431 |
+
\end{tabular}
|
| 432 |
+
\end{center}
|
| 433 |
+
|
| 434 |
+
\textbf{Novel biology.} Derm2 sits immediately upstream of Derm10 in Dingwall's Fig-5 Slingshot pseudotime, and Dingwall's CellChat analysis (Data S2) shows Derm2 sends Bmp5 and Rspo2 signals to epidermal placodes (Epi0/3/5), consistent with a signalling-competent Primary-EDEN precursor. Dingwall hypothesises this direction via Slingshot but does not statistically test Derm2 for En1-cKO depletion; our analysis extends En1-dependence to Derm2 at a modest but well-powered effect size ($1.25\times$, Fisher $p = 9.84\!\times\!10^{-5}$ at $n=1{,}586$). {\sloppy Artefacts: \pathsplit{discovery/pan\_skin/marker/100\_primary\_eden\_discovery.csv}, \pathsplit{100\_primary\_eden\_summary.json}, \pathsplit{101\_derm\_identity\_summary.json}, \pathsplit{101\_derm\_subcluster\_scores.csv}.\par}
|
| 435 |
+
|
| 436 |
+
\subsection{Melanoblast Eda-derepression: MITF axis vs neural-crest}
|
| 437 |
+
\label{sec:dingwall-melanoblast-mitf}
|
| 438 |
+
|
| 439 |
+
The melanoblast Eda-derepression phenotype is MITF-axis-driven rather than a neural-crest reversion. The pan-skin melanoblast prediction on Dingwall totals 5{,}143 cells (2{,}108 cKO / 3{,}035 WT). At baseline, cKO melanoblasts show two directionally-consistent pathway shifts relative to WT melanoblasts: Sweat\_gland module $\Delta = +0.021$ ($p = 1.0 \times 10^{-8}$) and Eda\_ectodysplasin $\Delta = +0.044$ ($p = 2.4 \times 10^{-6}$), i.e.\ the ectopic sweat-gland / ectodysplasin program leaks into the melanocyte lineage on \emph{En1} loss. This raises a mechanistic question: are the derepressing cKO melanoblasts \emph{reverting toward a neural-crest / bipotent progenitor state}, or is the melanocyte identity being \emph{dismantled at the MITF master-regulator axis} while the cell keeps its melanoblast identity?
|
| 440 |
+
|
| 441 |
+
To distinguish, we split cKO melanoblasts into the top and bottom quartile of within-cKO Eda-derepression score ($n = 527$ each) and computed the delta for three orthogonal modules: Neural\_crest, Melanogenesis\_late, and MITF\_regulon. The top-derepression quartile shows Neural\_crest $\Delta = +0.016$ ($p = 0.50$, ns), Melanogenesis\_late $\Delta = -0.017$ ($p = 0.47$, ns), and \textbf{MITF\_regulon $\mathbf{\Delta = -0.052}$} \textbf{($\mathbf{p = 1.2 \times 10^{-4}}$)} --- a significant but small decrement of the MITF regulon inside the derepressing subset, with no compensatory gain of neural-crest identity.
|
| 442 |
+
|
| 443 |
+
\textbf{Novel biology.} cKO melanoblasts that derepress the eccrine program show a small but significant MITF-regulon decrement with no compensatory neural-crest gain, consistent with the melanoblast identity being partially dismantled at the MITF axis rather than reverting to a bipotent neural-crest state. {\sloppy Artefacts: \pathsplit{discovery/pan\_skin/marker/106\_melanoblast\_nc\_summary.json}, \pathsplit{106\_melanoblast\_nc\_scores.csv}; script \pathsplit{scripts/analysis/106\_melanoblast\_neural\_crest.py}.\par}
|
| 444 |
+
|
| 445 |
+
\subsection{Per-class DEGs under En1-cKO}
|
| 446 |
+
\label{sec:dingwall-hf-placode-degs}
|
| 447 |
+
|
| 448 |
+
HF-placode is the most transcriptionally perturbed pan-skin class under En1-cKO. Per-PANDA-class Wilcoxon DE (cKO vs.\ WT) across all 12 non-trivial predicted skin classes on Dingwall, thresholded at $|\text{LFC}| > 1$ and $p_\text{adj} < 0.05$, ranks the classes by count of significantly differentially expressed genes:
|
| 449 |
+
|
| 450 |
+
\begin{center}
|
| 451 |
+
\small
|
| 452 |
+
\setlength{\tabcolsep}{4pt}
|
| 453 |
+
\begin{tabular}{p{4.6cm}p{2.2cm}p{8.0cm}}
|
| 454 |
+
\toprule
|
| 455 |
+
Predicted class & DEGs & Top genes \\
|
| 456 |
+
\midrule
|
| 457 |
+
\textbf{HF-placode} & \textbf{12} (all up) & Mybpc1, Bmpr1b, Tnc, Esr1, Meis2, Ccdc3, Kcnh7, Cntn5, Pcdh9 \\
|
| 458 |
+
melanoblast & 8 (7 up, 1 down) & Mybpc1, Bmpr1b, \emph{En1}, Kcnh7, Ttn, Esr1 \\
|
| 459 |
+
fibroblast-reticular & 4 (all up) & Mybpc1, Ttn, Krt5 \\
|
| 460 |
+
spinous & 3 (2 up, 1 down) & Slc1a3; \emph{Acer3} down \\
|
| 461 |
+
HF-ORS & 2 & Mybpc1 \\
|
| 462 |
+
basal-IFE & 2 & Meis2 \\
|
| 463 |
+
melanocyte & 2 & Bmpr1b \\
|
| 464 |
+
endothelial, immune, granular, mel-precursor, sebaceous & $\leq 1$ & --- \\
|
| 465 |
+
\bottomrule
|
| 466 |
+
\end{tabular}
|
| 467 |
+
\end{center}
|
| 468 |
+
|
| 469 |
+
\textbf{Novel biology.} The most transcriptionally perturbed pan-skin class under \emph{En1}-cKO is \textbf{HF-placode}, not the sweat-gland-fated compartments the canonical eccrine-specification story would predict. All 12 HF-placode DEGs are up in cKO (consistent with derepression) and include Bmpr1b (placode-induction BMP receptor), Tnc (placode ECM), Esr1, and Meis2. The largest molecular footprint of \emph{En1} loss lands on placode-fated ectoderm rather than the sweat-gland-committed lineage, so \emph{En1}'s spatial-repressor activity acts at least as strongly on induction as on commitment. {\sloppy Artefacts: \pathsplit{discovery/pan\_skin/marker/107\_dingwall\_class\_deg\_count.csv} and \pathsplit{107\_dingwall\_class\_deg\_count.json}; script \pathsplit{scripts/analysis/107\_dingwall\_class\_deg\_count.py}.\par}
|
| 470 |
+
|
| 471 |
+
\subsection{Synthesis and testable predictions}
|
| 472 |
+
|
| 473 |
+
Taken together, En1 acts as a \textbf{spatial repressor} of the ectodermal-appendage / placode / eccrine signalling program: Eda derepression is spatially distributed across HF-placode, dermis, and melanoblast compartments (\S\ref{sec:dingwall-pathway}); the melanoblast lineage additionally acquires ectopic Sweat\_gland and BMP signal; reticular dermis acquires a Basal\_keratinocyte signal; class-composition shifts (\S\ref{sec:dingwall-fisher}) show strong spinous depletion with placode and melanocyte enrichment; and the Primary-EDEN candidate Derm2 (\S\ref{sec:dingwall-primary-eden}) extends the phenotype upstream in the paper's Fig-5 lineage. This model yields the following wet-lab predictions:
|
| 474 |
+
|
| 475 |
+
\begin{enumerate}
|
| 476 |
+
\item \textbf{En1 as spatial repressor}: ISH for Foxi3/Muc5b/Krt8 will show broad-weak expression across cKO volar epidermis vs.\ discrete-strong expression at WT placode sites.
|
| 477 |
+
\item \textbf{Dermal Eda upregulation}: ISH on sorted cKO reticular fibroblasts will show elevated \emph{Eda}.
|
| 478 |
+
\item \textbf{Melanoblast-lineage ectopic BMP/sweat}: pSMAD1/5 staining in Dct$^+$ melanoblasts will show elevated signal in cKO paws.
|
| 479 |
+
\item \textbf{Spinous-layer differentiation defect}: Krt10 IHC on cKO volar skin will show reduced spinous-layer thickness relative to WT.
|
| 480 |
+
\item \textbf{Derm2 depletion}: FISH combining Enpp2 $+$ Bmp5 $+$ Rspo2 co-expression on P2.5 dermal sections will show reduced signal in cKO relative to WT.
|
| 481 |
+
\end{enumerate}
|
| 482 |
+
|
| 483 |
+
\section{Discovery target: Dahlin Kit-mutant hematopoiesis}
|
| 484 |
+
\label{sec:dahlin}
|
| 485 |
+
|
| 486 |
+
\subsection{Zero-shot classification recovers Dahlin's compositional shifts}
|
| 487 |
+
|
| 488 |
+
Held-out target: Dahlin \emph{et al.}\ 2018 \cite{dahlin2018kit} (Wilson lab), GSE107727. After per-sample QC (min\_counts $>1000$, min\_genes $>500$, mt\_frac $<0.10$; ENSMUSG$\to$symbol retains 27{,}044/27{,}998 features), 61{,}122 LSK/Kit$^+$ HSPCs recovered: WT $n=46{,}447$, c-Kit W41/W41 $n=14{,}675$. Evaluated on Dahlin with no cell-level overlap; 80.4\% HVG overlap after symbol conversion.
|
| 489 |
+
|
| 490 |
+
Fisher exact class enrichment (Kit-mutant vs.\ WT):
|
| 491 |
+
\begin{center}
|
| 492 |
+
\begin{tabular}{lrrrrl}
|
| 493 |
+
\toprule
|
| 494 |
+
Class & Kit\_W41 \% & WT \% & $\log_2$ fc & Fisher $p$ & Paper claim \\
|
| 495 |
+
\midrule
|
| 496 |
+
MPP & 53.6 & 83.5 & $-0.64$ & $\approx 0$ & Smaller HSC pool $\checkmark$ \\
|
| 497 |
+
erythroid & 42.6 & 13.3 & $+1.68$ & $\approx 0$ & Expanded erythroid $\checkmark$ \\
|
| 498 |
+
myeloid & 1.7 & 0.4 & $+2.06$ & $3.9 \times 10^{-48}$ & Expanded myeloid $\checkmark$ \\
|
| 499 |
+
lymphoid & 0.4 & 0.9 & $-1.02$ & $1.2 \times 10^{-8}$ & Global shift $\checkmark$ \\
|
| 500 |
+
megakaryocyte & 1.8 & 1.9 & $-0.07$ & $0.48$ & --- \\
|
| 501 |
+
\bottomrule
|
| 502 |
+
\end{tabular}
|
| 503 |
+
\end{center}
|
| 504 |
+
|
| 505 |
+
Every directional compositional shift the Dahlin paper reports for the c-Kit W41 mutant is reproduced by \panda{} at $p \approx 0$.
|
| 506 |
+
|
| 507 |
+
\subsection{Within-class pathway module analysis}
|
| 508 |
+
\label{sec:dahlin-pathway}
|
| 509 |
+
|
| 510 |
+
Pathway module analysis extends the panel to 30 modules per system (canonical marker programs + LT\_HSC\_quiescence, Erythroid\_dev, Granulopoiesis, Lymphopoiesis\_B/T, OXPHOS, Glycolysis, Redox\_glutathione, Ribosomal, Autophagy, Apoptosis\_pro/anti, ISR, Kit\_signaling, and more). Top Kit-W41 vs.\ WT within-class hits on Dahlin:
|
| 511 |
+
|
| 512 |
+
\begin{center}
|
| 513 |
+
\small
|
| 514 |
+
\setlength{\tabcolsep}{4pt}
|
| 515 |
+
\begin{tabular}{p{2.3cm}p{3.4cm}p{1.5cm}p{2.2cm}p{5.2cm}}
|
| 516 |
+
\toprule
|
| 517 |
+
Class & Module & $\Delta$ (Kit $-$ WT) & MannU $p_\text{adj}$ & Interpretation \\
|
| 518 |
+
\midrule
|
| 519 |
+
\textbf{MPP} & \textbf{Redox\_glutathione} & $\mathbf{+0.059}$ & $\mathbf{1.2 \times 10^{-127}}$ & Redox stress lead finding \\
|
| 520 |
+
MPP & Apoptosis\_pro & $+0.067$ & $\mathbf{5.2 \times 10^{-125}}$ & Compensatory pro-apoptosis \\
|
| 521 |
+
megakaryocyte & LT\_HSC\_quiescence & $-0.085$ & $\mathbf{3.2 \times 10^{-104}}$ & Loss of quiescence signature \\
|
| 522 |
+
erythroid & Glycolysis & $+0.051$ & $\mathbf{4.0 \times 10^{-95}}$ & Metabolic reprogramming \\
|
| 523 |
+
MPP & Integrated\_stress & $+0.021$ & $\mathbf{1.4 \times 10^{-28}}$ & ISR upregulation \\
|
| 524 |
+
MPP & Kit\_signaling & $-0.271$ & $\approx 0$ & Positive control (expected from Kit W41 mutation) \\
|
| 525 |
+
erythroid & Apoptosis\_pro & $-0.017$ & $\mathbf{2.3 \times 10^{-13}}$ & Erythroid apoptosis suppression (Dahlin's central molecular claim) \\
|
| 526 |
+
\bottomrule
|
| 527 |
+
\end{tabular}
|
| 528 |
+
\end{center}
|
| 529 |
+
|
| 530 |
+
\textbf{Novel Kit-W41 phenotypes.} MPP Redox\_glutathione upregulation ($p_\text{adj} = 1.2 \times 10^{-127}$) is the single most-significant hit --- a pathway the original Dahlin paper does not report, consistent with Kit-signalling loss compromising the redox buffering system that Sca1$^+$Kit$^+$ progenitors use to sustain quiescence. Megakaryocyte LT\_HSC\_quiescence loss ($p_\text{adj} = 3.2 \times 10^{-104}$) and erythroid Glycolysis $\uparrow$ ($p_\text{adj} = 4.0 \times 10^{-95}$) extend the metabolic-reprogramming picture. MPP Kit\_signaling collapse ($\Delta = -0.271$, $p \approx 0$) is the expected positive control from the c-Kit W41 loss-of-function itself. Dahlin's central erythroid Apoptosis\_pro $\downarrow$ claim is reproduced at $p_\text{adj} = 2.3 \times 10^{-13}$ (opposite direction from MPP's $+0.067$).
|
| 531 |
+
|
| 532 |
+
Artefacts: \texttt{discovery/hematopoiesis/marker/57\_pathway\_class\_by\_module\_padj.tsv} and \texttt{57\_pathway\_class\_by\_module\_delta.tsv}. The Kit\_signaling module collapse is visible in every predicted class (Figure~\ref{fig:dahlin}).
|
| 533 |
+
|
| 534 |
+
\begin{figure}[!htb]
|
| 535 |
+
\centering
|
| 536 |
+
\includegraphics[width=\textwidth]{figures/fig3_dahlin_heatmap.pdf}
|
| 537 |
+
\caption[Dahlin Kit-W41 vs.\ WT pathway heat-map]{Dahlin Kit-W41 vs.\ WT within-class pathway module scores. $\Delta = $ Kit\_W41 mean $-$ WT mean of per-cell module scores.\\
|
| 538 |
+
\textbf{Kit\_signaling:} collapses across every class ($\Delta = -0.27$ in MPP, $p_\text{adj} \approx 0$; expected positive control given the c-Kit W41 loss-of-function mutation).\\
|
| 539 |
+
\textbf{Erythroid Apoptosis\_pro $\downarrow$:} Dahlin's central claim ($p_\text{adj} = 2.3 \times 10^{-13}$; opposite direction from MPP $\Delta = +0.067$).\\
|
| 540 |
+
\textbf{MPP Redox\_glutathione $\uparrow$:} $p_\text{adj} = 1.2 \times 10^{-127}$.\\
|
| 541 |
+
\textbf{ISR upregulation:} $p_\text{adj} = 1.4 \times 10^{-28}$ in MPP.}
|
| 542 |
+
\label{fig:dahlin}
|
| 543 |
+
\end{figure}
|
| 544 |
+
|
| 545 |
+
\subsection{Dahlin marker deep-dive per predicted class}
|
| 546 |
+
\label{sec:dahlin-markers}
|
| 547 |
+
|
| 548 |
+
Full-class Wilcoxon per PANDA-predicted class on raw Dahlin counts (\pathsplit{discovery/hematopoiesis/marker/92\_dahlin\_marker\_deep\_dive.csv}). Current 10-class prediction distribution:
|
| 549 |
+
|
| 550 |
+
\begin{center}
|
| 551 |
+
\scriptsize
|
| 552 |
+
\setlength{\tabcolsep}{3pt}
|
| 553 |
+
\begin{tabular}{p{2.1cm}rrrrp{3.1cm}p{2.5cm}}
|
| 554 |
+
\toprule
|
| 555 |
+
Class & $n$ & WT & Kit\_W41 & frac\_WT & Top markers & Canonical recovery \\
|
| 556 |
+
\midrule
|
| 557 |
+
MPP & 31{,}754 & 27{,}931 & 3{,}823 & 0.880 & Cd34, Adgrl4, Sox4, Pim1 & 1/3 Kit-sig (Sox4) \\
|
| 558 |
+
erythroid & 15{,}397 & 8{,}352 & 7{,}045 & 0.542 & Ca1, Blvrb, Klf1, Aqp1 & 2/7 (Klf1, Blvrb) \\
|
| 559 |
+
megakaryocyte & 8{,}180 & 6{,}596 & 1{,}584 & 0.806 & Itga2b, Pbx1, Apoe, Gata2 & 2/4 (Itga2b, Pf4-adj) \\
|
| 560 |
+
myeloid & 3{,}463 & 1{,}755 & 1{,}708 & 0.507 & Elane, Mpo, Prtn3, Ctsg & 4/8 (Elane, Mpo, Prtn3, Ctsg) \\
|
| 561 |
+
basophil-mast & 848 & 603 & 245 & 0.711 & Cpa3, Ms4a2, Csrp3, Hdc & 4/5 (Cpa3, Ms4a2, Gata2, Hdc) \\
|
| 562 |
+
lymphoid & 694 & 639 & 55 & 0.921 & Gimap1, Gimap6, B2m & --- \\
|
| 563 |
+
monocyte & 472 & 315 & 157 & 0.667 & Irf8, Ms4a6c, Tyrobp & 2/8 (Mpo, Ctsg) \\
|
| 564 |
+
\textbf{LT-HSC} & \textbf{160} & \textbf{135} & \textbf{25} & \textbf{0.844} & \textbf{Hlf, Mpl, Meis1} & \textbf{3/7 (Hlf, Meis1, Mecom)} \\
|
| 565 |
+
pro-B & 66 & 60 & 6 & 0.909 & Ebf1, Cd79a, Vpreb3 & 1/4 (Il7r-adj) \\
|
| 566 |
+
macrophage & 63 & 42 & 21 & 0.667 & Irf8, Lgals1, Tyrobp & 1/8 (Ctsg) \\
|
| 567 |
+
\bottomrule
|
| 568 |
+
\end{tabular}
|
| 569 |
+
\end{center}
|
| 570 |
+
|
| 571 |
+
Basophil-mast, megakaryocyte, erythroid, and myeloid predictions recover canonical panels at LFC $3$--$8$. Erythroid is the class most enriched in Kit-W41 (frac\_WT $0.542$ vs corpus baseline $\sim\!0.76$), consistent with Kit-W41 retaining the erythroid pool but shifting its transcriptome (per \S\ref{sec:dahlin-pathway}). The LT-HSC class recovers 160 cells (84.4\% WT) with canonical Hlf/Mpl/Meis1/Mecom markers --- a Kit-W41-depleted stem population predicted as a first-class label rather than surfaced via abstain.
|
| 572 |
+
|
| 573 |
+
\subsection{Per-lineage metabolic reprogramming under Kit-W41}
|
| 574 |
+
\label{sec:dahlin-per-lineage-metabolism}
|
| 575 |
+
|
| 576 |
+
LT-HSCs and lymphoid progenitors show the strongest OXPHOS upregulation under Kit-W41. We scored all 61{,}122 Dahlin cells on four canonical metabolic modules --- OXPHOS\_ETC, Glycolysis, Fatty\_acid\_oxidation, and Redox\_glutathione --- and computed per-lineage Kit-W41-minus-WT deltas. Ranking lineages by $\sum_m |\Delta_m|$ (total mobilised metabolic signal):
|
| 577 |
+
|
| 578 |
+
\begin{center}
|
| 579 |
+
\begin{tabular}{lrrrrr}
|
| 580 |
+
\toprule
|
| 581 |
+
Predicted class & $\Delta$ OXPHOS & $\Delta$ Glycolysis & $\Delta$ FAO & $\Delta$ Redox & $\sum |\Delta|$ \\
|
| 582 |
+
\midrule
|
| 583 |
+
macrophage & $+0.044$ & $+0.190$ & $-0.042$ & $+0.098$ & $0.374$ \\
|
| 584 |
+
\textbf{LT-HSC} & $\mathbf{+0.219}$ & $+0.024$ & $-0.035$ & $+0.058$ & $\mathbf{0.335}$ \\
|
| 585 |
+
\textbf{lymphoid} & $\mathbf{+0.196}$ & $+0.039$ & $-0.002$ & $+0.095$ & $\mathbf{0.332}$ \\
|
| 586 |
+
erythroid & $+0.115$ & $+0.051$ & $-0.060$ & $+0.072$ & $0.299$ \\
|
| 587 |
+
MPP & $+0.138$ & $+0.063$ & $-0.009$ & $+0.059$ & $0.269$ \\
|
| 588 |
+
megakaryocyte & $+0.100$ & $+0.016$ & $-0.012$ & $+0.057$ & $0.185$ \\
|
| 589 |
+
basophil-mast & $-0.036$ & $+0.058$ & $+0.016$ & $+0.037$ & $0.147$ \\
|
| 590 |
+
myeloid & $-0.008$ & $+0.073$ & $+0.014$ & $+0.026$ & $0.122$ \\
|
| 591 |
+
monocyte & $+0.023$ & $+0.013$ & $+0.019$ & $+0.027$ & $0.082$ \\
|
| 592 |
+
\bottomrule
|
| 593 |
+
\end{tabular}
|
| 594 |
+
\end{center}
|
| 595 |
+
|
| 596 |
+
\textbf{Novel biology.} Setting aside macrophage (small $n$, glycolysis-dominated), the two lineages with the strongest, most directionally-coherent metabolic reprogramming are \textbf{LT-HSC} (OXPHOS $\Delta = +0.219$) and \textbf{lymphoid} (OXPHOS $\Delta = +0.196$): both show large OXPHOS gains with modest Redox/Glycolysis gains and a slight FAO drop --- the signature of a shift from a quiescent, FAO-supported state to an ETC-active state. This is consistent with hypofunctional Kit forcing canonically-quiescent HSCs into an energetically active (and stressed) state, with the effect largest per cell in the LT-HSC compartment. This per-lineage magnitude ranking is not reported in Dahlin. {\sloppy Artefacts: \pathsplit{discovery/hematopoiesis/marker/108\_dahlin\_lineage\_metabolism\_ranked.csv} and \pathsplit{108\_dahlin\_lineage\_metabolism.json}; script \pathsplit{scripts/analysis/108\_dahlin\_lineage\_metabolism.py}.\par}
|
| 597 |
+
|
| 598 |
+
\subsection{Testable wet-lab predictions on Kit-mutant mice}
|
| 599 |
+
|
| 600 |
+
\begin{enumerate}
|
| 601 |
+
\item \textbf{Erythroid-restricted anti-apoptotic switch}: Bcl2 and Bcl2l1 (Bcl-xL) protein levels should be elevated in Kit-W41 erythroid progenitors (Ter119$^+$-gated) but not in Kit-W41 MPP (Lin$^-$Kit$^+$Sca1$^+$-gated), reflecting the direction-flip in Apoptosis\_pro module score ($\Delta = -0.017$ in erythroid vs.\ $\Delta = +0.067$ in MPP).
|
| 602 |
+
\item \textbf{Integrated stress response as a druggable node}: ISRIB or GADD34 hyperactivator treatment of Kit-W41 mice should partially rescue the proliferation defect. ATF4$^+$ nuclei should be elevated in Kit-W41 bone-marrow sections across MPP, erythroid, myeloid, and megakaryocyte compartments.
|
| 603 |
+
\item \textbf{Redox rescue}: Supplementation with N-acetylcysteine or reduced glutathione should partially rescue the MPP Redox\_glutathione $\uparrow$ phenotype, providing a druggable node not previously implicated in Kit-W41 biology.
|
| 604 |
+
\item \textbf{Compensatory Kit-independent proliferation pathway}: because Kit\_signaling collapses across every class ($p \approx 0$) while some proliferation persists, an alternative RTK (Flt3, Csf1r) or non-RTK pathway must be compensating.
|
| 605 |
+
\end{enumerate}
|
| 606 |
+
|
| 607 |
+
\section{Discovery target: Veres SC-$\beta$ protocol imperfection}
|
| 608 |
+
\label{sec:veres}
|
| 609 |
+
|
| 610 |
+
\subsection{Zero-shot classification on Veres hPSC differentiation}
|
| 611 |
+
|
| 612 |
+
Veres \emph{et al.}\ 2019 \cite{veres2019scbeta}, \emph{Nature} 569:368--373 (Melton lab), GSE114412: hPSC-directed pancreatic differentiation across Stages 3--6 (foregut endoderm $\to$ stem-cell $\beta$, SC-$\beta$), 69{,}594 total cells; 57{,}297 in training, 12{,}297 held out (\S\ref{sec:zeroshot-veres}). Held-out macro F1 $= 0.900$ (Marker) on 20 canonical Veres classes; 82.8\% HVG overlap after human$\to$mouse case-fold.
|
| 613 |
+
|
| 614 |
+
Predicted class distribution across Veres stages (n=9{,}080 cells with parseable stage annotation; the remaining 3,217 cells are primary-islet controls without a stage tag):
|
| 615 |
+
\begin{center}
|
| 616 |
+
\small
|
| 617 |
+
\setlength{\tabcolsep}{5pt}
|
| 618 |
+
\resizebox{\textwidth}{!}{%
|
| 619 |
+
\begin{tabular}{p{4.4cm}rrrr}
|
| 620 |
+
\toprule
|
| 621 |
+
Predicted class & Stage 3 (n=2{,}899) & Stage 4 (n=2{,}577) & Stage 5 (n=1{,}589) & \textbf{Stage 6 (n=2{,}015)} \\
|
| 622 |
+
\midrule
|
| 623 |
+
pancreatic-progenitor & 72\% & 0\% & 0\% & 0\% \\
|
| 624 |
+
proliferating & 28\% & 14\% & 5\% & 2\% \\
|
| 625 |
+
endocrine-progenitor-primed & 0\% & 41\% & 0\% & 0\% \\
|
| 626 |
+
alpha\_progenitor & 0\% & 26\% & 49\% & \textbf{59\%} \\
|
| 627 |
+
beta\_progenitor & 0\% & 0\% & 11\% & \textbf{10\%} \\
|
| 628 |
+
epsilon & 0\% & 0\% & 13\% & 11\% \\
|
| 629 |
+
exocrine & 0\% & 0\% & 14\% & 10\% \\
|
| 630 |
+
ductal, delta & 0\% & 0\% & 3\% & 4\% each \\
|
| 631 |
+
\bottomrule
|
| 632 |
+
\end{tabular}}
|
| 633 |
+
\end{center}
|
| 634 |
+
|
| 635 |
+
\texttt{pancreatic-progenitor} dominates Stage 3 (72\%) and collapses across Stages 4--5 as endocrine-progenitor-primed and alpha\_progenitor identities emerge; by Stage 6, alpha\_progenitor is the largest class (59\%) and beta\_progenitor is 10\%. Adult-$\alpha$/adult-$\beta$ sub-prototypes never activate; SC-$\beta$-protocol cells occupy immature progenitor identities in the corpus manifold. Without any stage information supplied as input, \panda{} reproduces Veres's central finding that the differentiation is imperfect --- alpha-lineage outnumbers beta-lineage $\sim\!6\times$ at the SC-$\beta$ target stage (Figure~\ref{fig:veres}).
|
| 636 |
+
|
| 637 |
+
\begin{figure}[!htb]
|
| 638 |
+
\centering
|
| 639 |
+
\includegraphics[width=\textwidth]{figures/fig4_veres_stage_stack.pdf}
|
| 640 |
+
\caption[Veres 2019 stage-stack composition]{Veres 2019 iPSC-directed pancreatic differentiation. Stacked bars show PANDA-Marker predicted class fraction across the four staged Veres samples ($n=9{,}080$ cells with parseable stage annotation).\\
|
| 641 |
+
\textbf{Stage 3:} \texttt{pancreatic-progenitor} dominates (72\% of $n=2{,}899$) and monotonically declines through the protocol.\\
|
| 642 |
+
\textbf{Stage 6:} \texttt{alpha\_progenitor} rises from 0\% at Stage 3 to 59\% at the SC-$\beta$ target stage.\\
|
| 643 |
+
\textbf{Sub-prototypes:} Adult-$\beta$ and adult-$\alpha$ sub-prototypes do not activate on Veres (see \S\ref{sec:zeroshot-veres}).}
|
| 644 |
+
\label{fig:veres}
|
| 645 |
+
\end{figure}
|
| 646 |
+
|
| 647 |
+
\subsection{Stage-6 alpha\_progenitor vs beta\_progenitor mechanism}
|
| 648 |
+
\label{sec:veres-stage6-mech}
|
| 649 |
+
|
| 650 |
+
Restricting to Stage 6: $1{,}192$ alpha\_progenitor vs $199$ beta\_progenitor predictions. The marker deep-dive (\S\ref{sec:veres-marker-deep-dive}) places the identity boundary on the master-TF axis: alpha\_progenitor recovers 4/4 canonical adult-$\alpha$ markers (GCG, ARX, IRX2, MAFB) at LFC $+3$ to $+7.5$; beta\_progenitor top-3 are ACVR1C ($+4.5$), CALB2 ($+5.9$), INS ($+4.8$). The residual $6\times$ alpha/beta asymmetry at Stage 6 matches Veres's own report of SC-$\beta$-protocol inefficiency. Because the corpus adult-$\beta$ prototype does not activate on Veres (\S\ref{sec:veres-adult-beta}), Stage-6 secretory cells route to \texttt{beta\_progenitor} rather than \texttt{beta}, and immature-$\beta$ cannot be distinguished from committed adult-$\beta$ on this checkpoint.
|
| 651 |
+
|
| 652 |
+
\subsection{Veres marker deep-dive per predicted class}
|
| 653 |
+
\label{sec:veres-marker-deep-dive}
|
| 654 |
+
|
| 655 |
+
Wilcoxon per PANDA-predicted class on raw Veres counts (\pathsplit{discovery/pancreas/marker/91\_veres\_marker\_deep\_dive.csv}). Top-9 classes by cell count shown; distribution reflects the current 18-class predicted vocabulary on the held-out slice:
|
| 656 |
+
\begin{center}
|
| 657 |
+
\scriptsize
|
| 658 |
+
\setlength{\tabcolsep}{3pt}
|
| 659 |
+
\begin{tabular}{p{3.0cm}rp{4.8cm}p{2.6cm}r}
|
| 660 |
+
\toprule
|
| 661 |
+
Predicted class & $n$ & Top-3 markers (LFC) & Canonical panel recovery & \% of class at Stage 6 \\
|
| 662 |
+
\midrule
|
| 663 |
+
alpha\_progenitor & 2{,}864 & GCG ($+7.5$), TTR ($+4.9$), CHGA ($+4.7$) & 4/4 alpha (GCG, ARX, IRX2, MAFB) & 42\% \\
|
| 664 |
+
pancreatic-progenitor & 2{,}099 & MDK ($+3.5$), SOX11 ($+3.9$), FN1 ($+4.5$) & --- & 0\% \\
|
| 665 |
+
proliferating & 1{,}273 & TUBA1B ($+2.7$), TUBB ($+2.5$), HMGB1 ($+2.2$) & --- & 3\% \\
|
| 666 |
+
endocrine-progenitor-primed & 1{,}057 & DLK1 ($+6.0$), LDHB ($+3.0$), PDX1 ($+2.9$) & 1/5 beta (PDX1) & 0\% \\
|
| 667 |
+
epsilon & 701 & DDC ($+5.7$), FEV ($+5.1$), CHGA ($+5.3$) & 2/2 EP-Fev (FEV, INSM1) & 31\% \\
|
| 668 |
+
beta & 560 & INS ($+7.4$), IAPP ($+9.6$), ADCYAP1 ($+7.5$) & 1/5 beta (INS) $+$ MAFA/UCN3 in top-15 & 0\% \\
|
| 669 |
+
beta\_progenitor & 539 & ACVR1C ($+4.5$), CALB2 ($+5.9$), INS ($+4.8$) & 1/2 EP-Fev (INSM1) & 37\% \\
|
| 670 |
+
delta & 533 & SST ($+7.0$), ISL1 ($+2.6$), PCP4 ($+2.7$) & 1/2 EP-Fev (INSM1) & 16\% \\
|
| 671 |
+
acinar & 426 & REG1A ($+11.8$), CTRB1 ($+12.2$), CTRB2 ($+11.8$) & 4/4 acinar (PRSS1, PRSS2, CEL, CTRB1) & 0\% \\
|
| 672 |
+
\bottomrule
|
| 673 |
+
\end{tabular}
|
| 674 |
+
\end{center}
|
| 675 |
+
|
| 676 |
+
\textbf{Interpretation.} On the current pan-pancreatic checkpoint, \panda{}-Marker does not separate adult-$\alpha$ from juvenile $\alpha$ or adult-$\beta$ from juvenile $\beta$ on the Veres held-out slice. The \texttt{alpha\_progenitor} cluster recovers all four canonical adult-$\alpha$ markers (GCG, ARX, IRX2, MAFB) and dominates Veres Stage 6 (42\%). Acinar recovers all four PRSS1/PRSS2/CTRB1/CEL exocrine markers at LFC $\sim 10$.
|
| 677 |
+
|
| 678 |
+
\paragraph{Adult-$\beta$ signal within the \texttt{beta} class.}\label{sec:veres-adult-beta}
|
| 679 |
+
\panda{}-Marker predicts 0 Veres cells as \texttt{adult-beta} on the current checkpoint; the adult sub-prototype does not activate. The adult-maturity signature nonetheless survives at the gene level inside the \texttt{beta} class ($n=560$): mean-\texttt{log1p} enrichment vs the whole Veres dataset is MAFA $16\times$ (0.89 vs 0.06), UCN3 $8\times$, IAPP $12\times$, INS $3\times$. Two consistent readings: (i) Veres SC-$\beta$ cells occupy the corpus \texttt{beta} manifold too broadly for the adult sub-prototype's angular margin to fire; (ii) the corpus \texttt{adult-beta} prototype was learned on primary human islet data and the SC-$\beta$ transcriptome is close-but-not-close-enough. Artefact \texttt{95\_adult\_beta\_validation.json} is marked \texttt{vacuous: true} for the adult-vs-juvenile ratio.
|
| 680 |
+
|
| 681 |
+
\subsection{Polyhormonal SC-$\alpha$ sub-clustering}
|
| 682 |
+
\label{sec:veres-polyhormonal-subcluster}
|
| 683 |
+
|
| 684 |
+
Polyhormonal SC-$\alpha$ is a discrete sub-cluster (cluster 3), not stochastic co-expression across the pool. Veres 2019 reports that the SC-$\alpha$ output is polyhormonal (Gcg + Ins + Sst co-expression) but does not resolve whether this reflects a single trapped-bipotent sub-population or noise-floor stochastic co-expression across the SC-$\alpha$ pool. To test this we pooled Veres alpha-lineage predictions (\emph{alpha\_progenitor} + \emph{alpha}, $n = 3{,}473$), defined \emph{polyhormonal} as $\geq 2$ of \{Ins, Gcg, Sst\} above their respective 75\textsuperscript{th} percentiles within the alpha pool, and Leiden-clustered at resolution $= 0.5$ into 8 sub-clusters:
|
| 685 |
+
|
| 686 |
+
\begin{center}
|
| 687 |
+
\begin{tabular}{lrrrrrr}
|
| 688 |
+
\toprule
|
| 689 |
+
Sub-cluster & $n$ & \% pool & \% polyhormonal & \% GCG-hi & \% SST-hi & Enrichment \\
|
| 690 |
+
\midrule
|
| 691 |
+
\textbf{3} & \textbf{522} & \textbf{15\%} & \textbf{46\%} & \textbf{69\%} & \textbf{43\%} & \textbf{$\mathbf{2.53\times}$} \\
|
| 692 |
+
0 & 868 & 25\% & 30\% & 43\% & 29\% & $1.65\times$ \\
|
| 693 |
+
2 & 585 & 17\% & 15\% & 14\% & 27\% & $0.82\times$ \\
|
| 694 |
+
6, 4, 1, 5, 7 & 1{,}498 & 43\% & $\leq 9\%$ & --- & --- & $\leq 0.48\times$ \\
|
| 695 |
+
\midrule
|
| 696 |
+
baseline & 3{,}473 & 100\% & 18.1\% & --- & --- & $1.00\times$ \\
|
| 697 |
+
\bottomrule
|
| 698 |
+
\end{tabular}
|
| 699 |
+
\end{center}
|
| 700 |
+
|
| 701 |
+
\textbf{Novel biology.} Cluster 3 is a discrete polyhormonal sub-cluster: 15\% of the alpha pool, 46\% polyhormonal (\emph{i.e.}\ $2.5\times$ enriched over the 18.1\% baseline), with 69\% GCG-hi, 43\% SST-hi, and 36\% INS-hi \emph{simultaneously} elevated. Cluster 0 is a milder polyhormonal shoulder ($1.65\times$), and sub-clusters 4--7 are essentially devoid of polyhormonal cells ($\leq 9\%$). Polyhormonal SC-$\alpha$ is thus not uniform stochastic co-expression scattered across the SC-$\alpha$ pool; it is concentrated in a distinct sub-cluster comprising roughly 15--40\% of the pool (cluster 3 alone, or clusters 3+0). This supports a "trapped bipotent progenitor" interpretation of the Veres polyhormonal phenotype over a "noise-floor" interpretation: cluster 3 is a candidate for FISH-sorting and downstream functional / lineage-tracing assays. {\sloppy Artefacts: \pathsplit{discovery/pancreas/marker/110\_veres\_polyhormonal\_alpha\_per\_cluster.csv} and \pathsplit{110\_veres\_polyhormonal\_alpha\_summary.json}; script \pathsplit{scripts/analysis/110\_veres\_polyhormonal\_alpha.py}.\par}
|
| 702 |
+
|
| 703 |
+
\subsection{Testable wet-lab predictions on Veres SC-$\beta$ differentiation}
|
| 704 |
+
|
| 705 |
+
\begin{enumerate}
|
| 706 |
+
\item \textbf{Terminal TF-switching rescues yield}: overexpressing Nkx6-1, Mnx1, or Neurod1 during Stage 5$\to$6 should redirect a fraction of the SC-$\alpha$ population to SC-$\beta$, because the alpha/beta boundary is driven by the TF axis rather than by secretion machinery ($p = 1.9 \times 10^{-87}$ for beta\_TFs difference).
|
| 707 |
+
\item \textbf{Arx knockdown at Stage 5}: an Arx knockdown at the Stage 5 $\to$ 6 transition should convert SC-$\alpha$ to SC-$\beta$ or delta ($\log_2$ fc $+2.71$ for Arx in alpha).
|
| 708 |
+
\item \textbf{Polyhormonal sorting}: single-cell FISH for Hhex or Sst on Stage 6 cells should identify a Gcg$^+$Sst$^+$ double-positive population matching the delta\_TFs UP-in-alpha signal ($\Delta = +0.41$, $p = 7.5 \times 10^{-43}$); Gcg$^+$Sst$^-$ vs.\ Gcg$^+$Sst$^+$ transplantation assays would test the SC-$\alpha$ functional-immaturity claim.
|
| 709 |
+
\item \textbf{Adult-$\beta$ marker validation}: a MAFA-reporter line applied to Stage 6 output should identify a MAFA$^\text{high}$/UCN3$^\text{high}$/IAPP$^\text{high}$ subpopulation within the \panda{}-\texttt{beta} predicted cells (where MAFA/UCN3/IAPP are $8$--$16\times$ enriched at the gene level), even though the pan-pancreatic corpus's \texttt{adult-beta} sub-prototype does not activate on Veres cells.
|
| 710 |
+
\end{enumerate}
|
| 711 |
+
|
| 712 |
+
\section{Prototype geometry}
|
| 713 |
+
\label{sec:proto-geom}
|
| 714 |
+
|
| 715 |
+
For each system, \panda{}'s $K$ class-mean prototype vectors in the 128-d projection space have a participation-ratio effective dimensionality
|
| 716 |
+
\[
|
| 717 |
+
\text{eff\_dim}(P) \;=\; \frac{\big(\sum_i \lambda_i\big)^2}{\sum_i \lambda_i^2},
|
| 718 |
+
\]
|
| 719 |
+
where $\{\lambda_i\}$ are the eigenvalues of the centred prototype covariance. This is a smooth ``how many independent directions is the prototype set using?'' summary that equals $K$ when prototypes are orthonormal and 1 when they are collinear:
|
| 720 |
+
|
| 721 |
+
\begin{center}
|
| 722 |
+
\begin{tabular}{lrrr}
|
| 723 |
+
\toprule
|
| 724 |
+
System & $K$ & Effective dim & Fraction \\
|
| 725 |
+
\midrule
|
| 726 |
+
Pan-skin & 13 & \textbf{11.08} & 85\% \\
|
| 727 |
+
Pan-hematopoietic & 15 & \textbf{11.33} & 76\% \\
|
| 728 |
+
Pan-pancreatic & 20 & \textbf{10.85} & 54\% \\
|
| 729 |
+
\bottomrule
|
| 730 |
+
\end{tabular}
|
| 731 |
+
\end{center}
|
| 732 |
+
|
| 733 |
+
\textbf{Interpretation.} The pan-skin prototype set uses $85\%$ of the available dimensions, indicating cleanly-separated identities that occupy nearly-orthogonal directions in the projection space. The pan-hematopoietic prototype set uses $76\%$: still healthy, but the shared myeloid/basophil/megakaryocyte Gata1/Gata2$^+$ progenitor program links several lineages onto a common axis. The pan-pancreatic prototype set uses only $54\%$ of its 20 dimensions --- the largest prototype-set compression in the paper --- consistent with the well-established secondary transition dogma of pancreatic endocrinogenesis, in which $\beta$/$\delta$/endocrine-progenitor/other cell fates share a strong Nkx6-1$^+$/Neurod1$^+$ transcriptional program until terminal Ins1$^\text{hi}$/Mafa$^\text{hi}$ maturation. This is a direct empirical readout of biological lineage compression in the training corpus: the prototype geometry \panda{} learns is determined by transcriptomic distinguishability between labels, so labels whose training cells are close in expression space end up with lower-dimensional prototype sets even if the canonical ontology says they are distinct cell types.
|
| 734 |
+
|
| 735 |
+
\section{Cross-system synthesis}
|
| 736 |
+
|
| 737 |
+
Three tissue systems, three discovery targets, three paper-central findings independently reproduced by \panda{}:
|
| 738 |
+
|
| 739 |
+
\begin{center}
|
| 740 |
+
\small
|
| 741 |
+
\begin{tabular}{p{2.4cm}p{3.9cm}p{8.6cm}}
|
| 742 |
+
\toprule
|
| 743 |
+
System & Held-out target & Paper claim reproduced \\
|
| 744 |
+
\midrule
|
| 745 |
+
Skin & Dingwall En1-cKO & Unified spatial-repressor model $+$ EDEN lineage extension \\
|
| 746 |
+
Hematopoiesis & Dahlin Kit-W41 & Every 4-of-4 paper compositional and molecular claims \\
|
| 747 |
+
Pancreas & Veres hPSC & Polyhormonal SC-$\alpha$ $+$ terminal-TF-axis mechanism \\
|
| 748 |
+
\bottomrule
|
| 749 |
+
\end{tabular}
|
| 750 |
+
\end{center}
|
| 751 |
+
|
| 752 |
+
\textbf{Common threads.} Across all three systems: (i) zero-shot classification followed by within-class Wilcoxon and pathway module scoring recovers canonical marker sets never supplied to the model (Cd34/Myc for MPP; Cpa3/Gata2/Hdc/Mcpt8 for basophil-mast; Sox10/Dct/Tyr/Pmel for melanocyte; Nkx6-1/Mnx1/Neurod1 for SC-$\beta$; Arx/Irx2 for SC-$\alpha$); (ii) the confidence-gate abstain mechanism flags biology the training corpus does not represent (LT-HSC-like Hlf$^+$/Cd34$^+$ cluster in Dahlin, foregut endoderm in Veres, Derm-lineage subclusters in Dingwall); (iii) between-condition module-score contrasts reproduce the source paper's mechanistic interpretations at Wilcoxon $p$-values exceeding textbook thresholds by many orders of magnitude.
|
| 753 |
+
|
| 754 |
+
\textbf{Divergences.} The three systems separate cleanly by prototype geometry (\S\ref{sec:proto-geom}): skin at $85\%$ effective-dim utilisation shows fully-resolved terminal identities; hematopoiesis at $76\%$ has moderate lineage-shared compression on the Gata2$^+$ multipotent progenitor axis; pancreas at $54\%$ has strong compression on the shared endocrine-progenitor program. This ordering is itself a testable biological statement: the transcriptomic distinguishability of terminal identities in a training corpus is directly readable off the prototype set's effective dimensionality. Applying this framework prospectively lets an atlas-builder decide, before training, whether adult-anchor cells (or additional developmental stages) are needed to unlock terminal-cell-type separability.
|
| 755 |
+
|
| 756 |
+
Figure~\ref{fig:multiumap} shows the projection geometry for the three discovery targets; the condition of interest induces local density shifts within a cluster-preserving embedding, consistent with the within-class contrasts reported above.
|
| 757 |
+
|
| 758 |
+
\begin{figure}[!htb]
|
| 759 |
+
\centering
|
| 760 |
+
\includegraphics[width=\textwidth]{figures/fig6_multi_umap.pdf}
|
| 761 |
+
\caption[Three-panel discovery-target UMAP]{Three discovery-target UMAPs in \panda{}'s 128-d projection space (8{,}000 cells subsampled per panel).\\
|
| 762 |
+
\textbf{(a)} Dingwall skin coloured by En1 genotype.\\
|
| 763 |
+
\textbf{(b)} Dahlin hematopoiesis coloured by Kit genotype.\\
|
| 764 |
+
\textbf{(c)} Veres pancreatic differentiation coloured by protocol stage (3--6).\\
|
| 765 |
+
\textbf{Observation:} between-condition/between-stage differences manifest as local density shifts within a shared, biologically-structured embedding.}
|
| 766 |
+
\label{fig:multiumap}
|
| 767 |
+
\end{figure}
|
| 768 |
+
|
| 769 |
+
\section{Discussion}
|
| 770 |
+
|
| 771 |
+
\panda{} delivers a compact PCA-only classifier trained on 100\%-paper-labeled corpora that validates within-corpus, transfers to labeled hold-outs (both strict zero-shot and held-out-slice), and supports mechanistic discovery via within-class Wilcoxon DE and pathway module scoring.
|
| 772 |
+
|
| 773 |
+
Three points about how the checkpoint is used in discovery. First, the Dingwall EDEN validation illustrates the intended workflow of a prototype-anchored classifier as a hypothesis-generation tool: the pan-tissue fibroblast prototype is calibrated broader than any narrow tissue-specific compartment, so an independent clustering (Line B) or target-native training pass (Line C) is required to resolve EDEN as a single class; once either is provided, the depletion phenotype is a learnable and reproducible property of the labelling rather than a clustering artefact, and sub-cluster panel-matching extends the phenotype to the untested Derm2 precursor. Second, expanded pathway scoring on Dahlin surfaces MPP Redox\_glutathione $\uparrow$ ($p = 1.2\!\times\!10^{-127}$) as a novel Kit-W41 phenotype not reported in the original paper, alongside a megakaryocyte quiescence-signature loss and the erythroid-vs-MPP direction-flip in Apoptosis\_pro. Third, on Veres the adult-$\beta$ sub-prototype does not activate on SC-$\beta$ cells even though the adult-maturity signature (MAFA, UCN3, IAPP, ADCYAP1) is $8$--$16\times$ enriched at the gene level inside the \texttt{beta} class --- a resolution limit of the current corpus that the prototype-geometry compression (54\% effective dim, \S\ref{sec:proto-geom}) makes quantitatively concrete.
|
| 774 |
+
|
| 775 |
+
Future work will extend the interpretability toolkit --- prototype--gene attribution, counterfactual single-gene knockout, and Hessian gene-gene interaction probes --- using the analytic PCA back-projection that \panda{}'s architecture makes exact.
|
| 776 |
+
|
| 777 |
+
\section{Limitations}
|
| 778 |
+
|
| 779 |
+
\begin{itemize}
|
| 780 |
+
\item Pan-tissue prototypes are calibrated for cross-dataset generality; narrow tissue-specific compartments like Dingwall's EDEN require either an independent clustering step (Line B) or Dingwall-native training (Line C) to be recovered as a single class.
|
| 781 |
+
\item Zero-shot transfer to mature terminal cell types requires developmentally-mature anchors in the training corpus (documented via the Baron test-half zero-shot performance and the pancreatic prototype-geometry compression).
|
| 782 |
+
\item Sparse-class predictions (HF-placode, HF-DP, gamma/immune on Baron test) are underpowered when the training-corpus support is small.
|
| 783 |
+
\item Novel-population claims from confidence-gated abstained cells are hypotheses; every mechanistic claim in the discovery reports requires wet-lab confirmation.
|
| 784 |
+
\item Fibroblast-papillary predictions in Dingwall show smooth-muscle contamination (Cald1, Myh11, Acta2), consistent with the class being conflated with myofibroblast identity in adult skin.
|
| 785 |
+
\item Cross-platform generalisation to Smart-seq2 (Nestorowa) remains a hard benchmark; a Smart-seq2 LT-HSC training anchor would be the natural next step.
|
| 786 |
+
\end{itemize}
|
| 787 |
+
|
| 788 |
+
\section{Reproducibility}
|
| 789 |
+
|
| 790 |
+
\textbf{Multiple-testing scope.} Pathway-module $p$-values reported in the class $\times$ module tables (\S\ref{sec:dingwall-pathway}, \S\ref{sec:dahlin-pathway}) are Bonferroni-corrected within-system across the full $n_\text{classes} \times n_\text{modules}$ family scanned per system (Dingwall: $12 \times 15+$ modules; Dahlin: $10 \times 30$ modules). Per-class Wilcoxon DEG $p$-values (\S\ref{sec:dingwall-hf-placode-degs}, \S\ref{sec:dahlin-markers}, and the Veres per-class marker deep-dive) use Benjamini--Hochberg per class. Fisher exact class enrichments (\S\ref{sec:dingwall-fisher}) use Benjamini--Hochberg across the classes tested per system.
|
| 791 |
+
|
| 792 |
+
All quantitative claims in this paper trace to a specific artefact:
|
| 793 |
+
\begingroup
|
| 794 |
+
\RaggedRight
|
| 795 |
+
\sloppy
|
| 796 |
+
\begin{itemize}\itemsep2pt
|
| 797 |
+
\item Model checkpoints: \pathsplit{checkpoints/\{system\}/\{variant\}/panda\_final.pt} where variant $\in \{$pca, marker$\}$
|
| 798 |
+
\item Marker channel gene lists: \pathsplit{panda/markers.yaml}
|
| 799 |
+
\item Corpus builders: \pathsplit{scripts/\{pan\_skin,hematopoiesis,pancreas\}/}; unified trainer: \pathsplit{scripts/common/train\_panda.py}; unified 5-fold CV driver: \pathsplit{scripts/common/cv\_holdout.py}; zero-shot inference: \pathsplit{scripts/common/run\_all\_zero\_shot.py}
|
| 800 |
+
\item External-label supplements: \pathsplit{data/external\_labels/\{dingwall\_supp,haensel,joost2016,mca,mia,byrnes,yu,baccin,melanocyte\_anchor\}/}
|
| 801 |
+
\item Per-configuration 5-fold CV (three seeds per system): \pathsplit{discovery/\{system\}/\{variant\}/cv\_5fold.json}, \pathsplit{cv\_5fold\_seed1.json}, \pathsplit{cv\_5fold\_seed2.json}
|
| 802 |
+
\item Zero-shot labeled targets:
|
| 803 |
+
\begin{itemize}\itemsep2pt
|
| 804 |
+
\item Baron test-half: \pathsplit{discovery/pancreas/\{pca,marker\}/baron\_summary.json} (current 20-class vocabulary; the paper body's Baron numbers are read from this file)
|
| 805 |
+
\item Veres held-out 12{,}297: \pathsplit{discovery/pancreas/\{pca,marker\}/veres\_summary.json}
|
| 806 |
+
\item Nestorowa: \pathsplit{discovery/hematopoiesis/\{pca,marker\}/nestorowa\_summary.json}
|
| 807 |
+
\item Sulic: \pathsplit{discovery/pan\_skin/\{pca,marker\}/sulic\_summary.json}
|
| 808 |
+
\item Belote: \pathsplit{discovery/pan\_skin/\{pca,marker\}/belote\_summary.json}
|
| 809 |
+
\end{itemize}
|
| 810 |
+
\item Adult-$\beta$ marker-panel validation on Veres: \pathsplit{discovery/pancreas/marker/95\_adult\_beta\_validation.json}
|
| 811 |
+
\item Expanded pathway module analysis (all systems): \pathsplit{scripts/analysis/57\_pathway\_analysis.py}, outputs at \pathsplit{discovery/\{system\}/marker/57\_pathway\_class\_by\_module\_\{padj,delta\}.tsv}
|
| 812 |
+
\item \textbf{Dingwall EDEN validation (three lines of evidence)}:
|
| 813 |
+
\begin{itemize}\itemsep2pt
|
| 814 |
+
\item Line A post-hoc scoring: \pathsplit{scripts/analysis/98\_eden\_posthoc\_detection.py}, output \pathsplit{discovery/pan\_skin/marker/98\_eden\_summary.json}
|
| 815 |
+
\item Line B scanpy reproduction: \pathsplit{scripts/analysis/103\_replicate\_dingwall\_seurat\_pipeline.py}, output \pathsplit{data/processed/dingwall\_replica/dingwall\_replica.h5ad}, \pathsplit{data/processed/dingwall\_replica/replica\_cluster\_20\_qc.json}, \pathsplit{replica\_marker\_matches.csv}
|
| 816 |
+
\item Line C PANDA on Derm labels: \pathsplit{scripts/analysis/104\_train\_on\_dingwall\_derm\_labels.py}, output \pathsplit{discovery/pan\_skin/marker/104\_dingwall\_derm\_summary.json} $+$ prediction/depletion CSVs
|
| 817 |
+
\end{itemize}
|
| 818 |
+
\item Primary EDEN Derm2 discovery: \pathsplit{scripts/analysis/100\_primary\_eden\_discovery.py}, \pathsplit{101\_primary\_eden\_derm\_scoring.py}; outputs \pathsplit{discovery/pan\_skin/marker/100\_primary\_eden\_discovery.csv}, \pathsplit{100\_primary\_eden\_summary.json}, \pathsplit{101\_derm\_identity\_summary.json}, \pathsplit{101\_derm\_subcluster\_scores.csv}
|
| 819 |
+
\item Dingwall other mechanistic outputs: \pathsplit{discovery/pan\_skin/marker/57\_pathway\_analysis.csv, 90\_dingwall\_marker\_deep\_dive.csv}
|
| 820 |
+
\item Dahlin Kit-mutant: \pathsplit{discovery/hematopoiesis/marker/92\_dahlin\_marker\_deep\_dive.csv, dahlin\_summary.json}
|
| 821 |
+
\item Veres marker deep-dive: \pathsplit{discovery/pancreas/marker/91\_veres\_marker\_deep\_dive.csv}
|
| 822 |
+
\end{itemize}
|
| 823 |
+
\endgroup
|
| 824 |
+
|
| 825 |
+
\begin{thebibliography}{99}
|
| 826 |
+
\bibitem{dingwall2024en1cko} Dingwall CB, \emph{et al.}\ (Aldea D, Kamberov YG \emph{corresp.}) (2024). ``Divergent developmental origins for the mammalian sweat gland lineage revealed by an En1-Cre knock-in.'' \emph{Developmental Cell} 59(1):20--32.e6 \url{https://doi.org/10.1016/j.devcel.2023.11.017}. GSE220977.
|
| 827 |
+
\bibitem{veres2019scbeta} Veres A, \emph{et al.}\ (Melton DA \emph{corresp.}) (2019). ``Charting cellular identity during human in vitro $\beta$-cell differentiation.'' \emph{Nature} 569:368--373 \url{https://doi.org/10.1038/s41586-019-1168-5}. GSE114412.
|
| 828 |
+
\bibitem{dahlin2018kit} Dahlin JS, \emph{et al.}\ (Wilson NK \emph{corresp.}) (2018). ``A single-cell hematopoietic landscape resolves 8 lineage trajectories and defects in Kit mutant mice.'' \emph{Blood} 131(21):e1--e11 \url{https://doi.org/10.1182/blood-2017-12-821413}. GSE107727.
|
| 829 |
+
\bibitem{haensel2020skin} Haensel D, \emph{et al.}\ (Annusver K \emph{et al.}) (2020). ``Defining epidermal basal cell states during skin homeostasis and wound healing using single-cell transcriptomics.'' \emph{Cell Reports} 30(11):3932--3947.e6 \url{https://doi.org/10.1016/j.celrep.2020.02.091}. GSE142471.
|
| 830 |
+
\bibitem{joost2016} Joost S, \emph{et al.}\ (2016). ``Single-cell transcriptomics reveals that differentiation and spatial signatures shape epidermal and hair follicle heterogeneity.'' \emph{Cell Systems} 3(3):221--237.e9 \url{https://doi.org/10.1016/j.cels.2016.08.010}. GSE67602.
|
| 831 |
+
\bibitem{belote2021} Belote RL, \emph{et al.}\ (2021). ``Human melanocyte development and melanoma dedifferentiation at single-cell resolution.'' \emph{Nature Cell Biology} 23(9):1035--1047 \url{https://doi.org/10.1038/s41556-021-00740-8}. GSE151091.
|
| 832 |
+
\bibitem{han2018mca} Han X, \emph{et al.}\ (Guo G \emph{corresp.}) (2018). ``Mapping the mouse cell atlas by Microwell-Seq.'' \emph{Cell} 172(5):1091--1107.e17 \url{https://doi.org/10.1016/j.cell.2018.02.001}. GSE108097.
|
| 833 |
+
\bibitem{tms2020} Tabula Muris Consortium (2020). ``A single-cell transcriptomic atlas characterizes ageing tissues in the mouse.'' \emph{Nature} 583:590--595 \url{https://doi.org/10.1038/s41586-020-2496-1}. GSE132042.
|
| 834 |
+
\bibitem{baccin2020} Baccin C, \emph{et al.}\ (2020). ``Combined single-cell and spatial transcriptomics reveal the molecular, cellular and spatial bone marrow niche organization.'' \emph{Nature Cell Biology} 22(1):38--48 \url{https://doi.org/10.1038/s41556-019-0439-6}. GSE122465.
|
| 835 |
+
\bibitem{bastidas2019} Bastidas-Ponce A, \emph{et al.}\ (Bakhti M, Lickert H) (2019). ``Comprehensive single cell mRNA profiling reveals a detailed roadmap for pancreatic endocrinogenesis.'' \emph{Development} 146(12):dev173849 \url{https://doi.org/10.1242/dev.173849}. GSE132188.
|
| 836 |
+
\bibitem{byrnes2018} Byrnes LE, \emph{et al.}\ (Sneddon JB \emph{corresp.}) (2018). ``Lineage dynamics of murine pancreatic development at single-cell resolution.'' \emph{Nature Communications} 9(1):3922 \url{https://doi.org/10.1038/s41467-018-06176-3}. GSE101099.
|
| 837 |
+
\bibitem{yu2021} Yu X, \emph{et al.}\ (2021). ``Single-cell RNA-seq of the developing pancreas identifies novel endocrine progenitors and endocrine subtypes.'' \emph{Cell Research} 31:669--686 \url{https://doi.org/10.1038/s41422-021-00509-6}. GSE139627.
|
| 838 |
+
\bibitem{hrovatin2023mia} Hrovatin K, \emph{et al.}\ (2023). ``Delineating mouse $\beta$-cell identity during lifetime and in diabetes with a single cell atlas.'' \emph{Nature Metabolism} 5:1615--1637 \url{https://doi.org/10.1038/s42255-023-00876-x}. GSE211796.
|
| 839 |
+
\end{thebibliography}
|
| 840 |
+
|
| 841 |
+
\clearpage
|
| 842 |
+
\subsection*{Supplement figure index}
|
| 843 |
+
\label{sec:supp-index}
|
| 844 |
+
|
| 845 |
+
Supporting figures are provided as PDFs under \pathsplit{figures/supplement/} (S\emph{n}) and \pathsplit{figures/biology/} (B\emph{n}). Each entry lists filename, one-line description, and the paper section that motivates it.
|
| 846 |
+
|
| 847 |
+
\begingroup
|
| 848 |
+
\RaggedRight
|
| 849 |
+
\small
|
| 850 |
+
\begin{itemize}\itemsep1pt
|
| 851 |
+
\item \textbf{S1} \pathsplit{01\_cv\_summary.pdf} --- Held-out 5-fold CV summary across the three systems (\S\ref{sec:multiseed}; Fig~\ref{fig:cv}).
|
| 852 |
+
\item \textbf{S2} \pathsplit{02\_per\_class\_f1.pdf} --- Per-class F1 bars for skin/HSC/pancreas (Fig~\ref{fig:cv}).
|
| 853 |
+
\item \textbf{S3} \pathsplit{03\_prototype\_cosine.pdf} --- Prototype-prototype cosine similarity matrices per system (\S\ref{sec:proto-geom}).
|
| 854 |
+
% S4 removed --- legacy training-trajectory figure contained stale class vocab and wrong K counts.
|
| 855 |
+
\item \textbf{S5} \pathsplit{05\_adversary\_purification.pdf} --- Dataset/depth-adversary AUROC vs.\ ramp step (curriculum diagnostics for \S\ref{sec:multiseed}).
|
| 856 |
+
\item \textbf{S6} \pathsplit{06\_cross\_system\_prototypes.pdf} --- Cross-system prototype-geometry comparison (\S\ref{sec:proto-geom}).
|
| 857 |
+
% S7 removed --- attribution-heatmap contained deprecated class labels and leaked cell-barcode strings as gene columns.
|
| 858 |
+
% S8 removed --- TF-enrichment used stale skin/pancreas class rosters.
|
| 859 |
+
% S9 removed --- KO-essentials contained deprecated class labels and leaked cell-barcode strings.
|
| 860 |
+
% S10 removed --- Hessian gene-pair figure contained stale classes and illegible pair labels.
|
| 861 |
+
\item \textbf{S11} \pathsplit{11\_novel\_populations.pdf} --- Abstained-cell clusters flagged as candidate novel populations (\S{Discussion}, cross-system).
|
| 862 |
+
% S12 removed --- Co-attention modules referenced deprecated classes (nascent-eccrine-gland, UNK, unassigned, mesenchymal).
|
| 863 |
+
\item \textbf{S13} \pathsplit{13\_dingwall\_umap.pdf} --- Full Dingwall UMAP by predicted class and genotype (\S\ref{sec:dingwall}; Fig~\ref{fig:umap}).
|
| 864 |
+
\item \textbf{S14} \pathsplit{14\_dahlin\_umap.pdf} --- Full Dahlin UMAP by predicted class and Kit genotype (\S\ref{sec:dahlin}).
|
| 865 |
+
\item \textbf{S15} \pathsplit{15\_veres\_umap.pdf} --- Full Veres UMAP by predicted class and stage (\S\ref{sec:veres}).
|
| 866 |
+
\item \textbf{S16} \pathsplit{16\_dingwall\_discovery.pdf} --- Dingwall discovery panel: EDEN, Derm2, HF-placode DEGs (\S\ref{sec:dingwall-eden}, \S\ref{sec:dingwall-primary-eden}, \S\ref{sec:dingwall-hf-placode-degs}).
|
| 867 |
+
\item \textbf{S17} \pathsplit{17\_dahlin\_discovery.pdf} --- Dahlin discovery panel: MPP redox, MK quiescence, per-lineage metabolism (\S\ref{sec:dahlin-pathway}, \S\ref{sec:dahlin-per-lineage-metabolism}).
|
| 868 |
+
\item \textbf{S18} \pathsplit{18\_veres\_discovery.pdf} --- Veres discovery panel: Stage-6 alpha/beta, polyhormonal sub-cluster (\S\ref{sec:veres-polyhormonal-subcluster}).
|
| 869 |
+
\item \textbf{S19} \pathsplit{19\_myeloid\_network.pdf} --- Myeloid marker co-expression network (context: Dahlin \S\ref{sec:dahlin-markers}).
|
| 870 |
+
\item \textbf{S20} \pathsplit{20\_placode\_wnt\_module.pdf} --- HF-placode / Wnt module score across genotypes (\S\ref{sec:dingwall-pathway}, \S\ref{sec:dingwall-hf-placode-degs}).
|
| 871 |
+
\item \textbf{S23} \pathsplit{23\_anchor\_delta\_recall.pdf} --- Anchor-vs-holdout recall $\Delta$ across the four held-out labeled targets.
|
| 872 |
+
\item \textbf{S24} \pathsplit{24\_pca\_vs\_marker\_umaps\_dingwall\_by\_genotype.pdf} --- PCA vs.\ Marker UMAP of Dingwall coloured by En1 genotype (Marker vs.\ PCA ablation; \S\ref{sec:dingwall}).
|
| 873 |
+
\item \textbf{S24b} \pathsplit{24b\_pca\_vs\_marker\_umaps\_dingwall\_by\_class.pdf} --- Same Dingwall PCA vs.\ Marker UMAP coloured by predicted class.
|
| 874 |
+
\item \textbf{S25} \pathsplit{25\_pca\_vs\_marker\_umaps\_dahlin\_by\_genotype.pdf} --- PCA vs.\ Marker UMAP of Dahlin coloured by Kit genotype (\S\ref{sec:dahlin}).
|
| 875 |
+
\item \textbf{S25b} \pathsplit{25b\_pca\_vs\_marker\_umaps\_dahlin\_by\_class.pdf} --- Same Dahlin PCA vs.\ Marker UMAP coloured by predicted class.
|
| 876 |
+
\item \textbf{S26} \pathsplit{26\_pca\_vs\_marker\_umaps\_veres\_by\_stage.pdf} --- PCA vs.\ Marker UMAP of Veres coloured by protocol stage (\S\ref{sec:veres}).
|
| 877 |
+
\item \textbf{S26b} \pathsplit{26b\_pca\_vs\_marker\_umaps\_veres\_by\_class.pdf} --- Same Veres PCA vs.\ Marker UMAP coloured by predicted class.
|
| 878 |
+
\item \textbf{S27} \pathsplit{27\_dingwall\_en1\_enrichment.pdf} --- En1-cKO enrichment per PANDA class with Fisher $p$ (\S\ref{sec:dingwall-fisher}).
|
| 879 |
+
\item \textbf{S28} \pathsplit{28\_melanocyte\_pathway\_modules.pdf} --- Melanocyte-lineage pathway module scores in Dingwall (\S\ref{sec:dingwall-melanoblast-mitf}).
|
| 880 |
+
\item \textbf{B1} \pathsplit{biology\_01\_dingwall\_umap.pdf} --- Publication-style Dingwall UMAP by predicted class (\S\ref{sec:dingwall}).
|
| 881 |
+
\item \textbf{B2} \pathsplit{biology\_02\_primary\_eden.pdf} --- Primary-EDEN Derm2 sub-cluster and marker overlap (\S\ref{sec:dingwall-primary-eden}).
|
| 882 |
+
\item \textbf{B3} \pathsplit{biology\_03\_melanoblast\_mitf.pdf} --- Melanoblast MITF-regulon vs.\ neural-crest scoring (\S\ref{sec:dingwall-melanoblast-mitf}).
|
| 883 |
+
\item \textbf{B4} \pathsplit{biology\_04\_dahlin\_metabolism.pdf} --- Per-lineage metabolic reprogramming heat-map on Dahlin (\S\ref{sec:dahlin-per-lineage-metabolism}).
|
| 884 |
+
\item \textbf{B5} \pathsplit{biology\_05\_dahlin\_composition.pdf} --- Compositional shifts under Kit-W41 (\S\ref{sec:dahlin}, Fisher table).
|
| 885 |
+
\item \textbf{B6} \pathsplit{biology\_06\_veres\_beta\_quadrant.pdf} --- Veres beta-class MAFA/UCN3/IAPP quadrant (\S\ref{sec:veres-adult-beta}).
|
| 886 |
+
\item \textbf{B7} \pathsplit{biology\_07\_veres\_polyhormonal.pdf} --- Veres polyhormonal SC-$\alpha$ sub-cluster 3 (\S\ref{sec:veres-polyhormonal-subcluster}).
|
| 887 |
+
\item \textbf{B8} \pathsplit{biology\_08\_prototype\_geometry.pdf} --- Cross-system prototype effective-dimensionality plot (\S\ref{sec:proto-geom}).
|
| 888 |
+
\end{itemize}
|
| 889 |
+
\endgroup
|
| 890 |
+
|
| 891 |
+
\end{document}
|