Spaces:
Sleeping
Sleeping
File size: 6,237 Bytes
0fff343 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 | """Pinned constants for the cBioPortal CRC dataset we depend on.
If cBioPortal renames a file or column, this is the single place to update.
"""
from pathlib import Path
STUDY = "coadread_tcga_pan_can_atlas_2018"
# Datahub LFS-resolved URL. The S3 tarball mirror returns 403 and the plain
# raw.githubusercontent.com URL returns 131-byte LFS pointers; this is the URL
# pattern that yields real content.
BASE_URL = (
"https://media.githubusercontent.com/media/cBioPortal/datahub/master/public/"
+ STUDY
)
# Human study page (for the manual-download fallback message).
STUDY_PAGE_URL = f"https://www.cbioportal.org/study/summary?id={STUDY}"
DATAHUB_BROWSE_URL = (
f"https://github.com/cBioPortal/datahub/tree/master/public/{STUDY}"
)
# Logical name -> (filename, required_for_chunk1).
# `data_mutations.txt` is ~340 MB and currently over the datahub's GitHub LFS
# budget; we treat it as optional. Chunk 1 doesn't need it (TMB comes from the
# clinical sample file). Later chunks that want per-variant calls will need to
# fetch it from cBioPortal directly.
FILES = {
"sample": ("data_clinical_sample.txt", True),
"patient": ("data_clinical_patient.txt", True),
"expression": ("data_mrna_seq_v2_rsem.txt", True),
"mutations": ("data_mutations.txt", False),
}
# Project paths.
REPO_ROOT = Path(__file__).resolve().parent.parent
RAW_DIR = REPO_ROOT / "data" / "raw"
PROCESSED_DIR = REPO_ROOT / "data" / "processed"
# Columns we depend on per file. build.py fails loudly if any are absent, listing
# the columns it *did* find.
REQUIRED_SAMPLE_COLS = [
"PATIENT_ID",
"SAMPLE_ID",
"ONCOTREE_CODE", # COAD vs READ
"MSI_SENSOR_SCORE", # continuous MSI score; thresholded into msi_status
"TMB_NONSYNONYMOUS",
]
REQUIRED_PATIENT_COLS = [
"PATIENT_ID",
"AGE",
"SEX",
"AJCC_PATHOLOGIC_TUMOR_STAGE",
"OS_STATUS",
"OS_MONTHS",
]
# MSI label derivation. Thresholds are taken verbatim from the suggestion in the
# cBioPortal data_clinical_sample.txt header:
# "MSI Score reported by MSIsensor. The suggested thresholds are
# MSI: >10, Indeterminate: 4-10 and MSS: <10."
# We use >= for the boundary.
MSI_SENSOR_HIGH = 10.0 # >= this => MSI-H
MSI_SENSOR_LOW = 4.0 # < this => MSS; in-between => Indeterminate
# ---------------------------------------------------------------------------
# HNSC (head & neck) — second dataset. Mirrors the CRC layout above.
# ---------------------------------------------------------------------------
#
# Study: hnsc_tcga_pan_can_atlas_2018 — TCGA HNSC PanCancer Atlas.
# HPV status field: depending on the cBioPortal release the called
# "HPV+/HPV-" lives EITHER in the sample clinical file (column
# ``HPV_STATUS``) or is derivable from the patient ``SUBTYPE`` (suffix
# ``_HPV+`` / ``_HPV-``). The build script tries both, in that order,
# and fails loudly if neither is present.
HNSC_STUDY = "hnsc_tcga_pan_can_atlas_2018"
HNSC_BASE_URL = (
"https://media.githubusercontent.com/media/cBioPortal/datahub/master/public/"
+ HNSC_STUDY
)
HNSC_STUDY_PAGE_URL = f"https://www.cbioportal.org/study/summary?id={HNSC_STUDY}"
HNSC_DATAHUB_BROWSE_URL = (
f"https://github.com/cBioPortal/datahub/tree/master/public/{HNSC_STUDY}"
)
HNSC_FILES = {
"sample": ("data_clinical_sample.txt", True),
"patient": ("data_clinical_patient.txt", True),
"expression": ("data_mrna_seq_v2_rsem.txt", True),
}
HNSC_RAW_DIR = REPO_ROOT / "data" / "raw_hnsc"
HNSC_PROCESSED_DIR = REPO_ROOT / "data" / "processed_hnsc"
# Columns we depend on. HPV_STATUS may be absent on some releases — see the
# fallback in build_hnsc.py.
HNSC_REQUIRED_SAMPLE_COLS = ["PATIENT_ID", "SAMPLE_ID"]
HNSC_REQUIRED_PATIENT_COLS = ["PATIENT_ID", "AGE", "SEX", "AJCC_PATHOLOGIC_TUMOR_STAGE"]
# Columns / value patterns the build script will look for, in priority order.
HNSC_HPV_STATUS_CANDIDATES_SAMPLE = ["HPV_STATUS", "HPV_STATUS_ISH", "HPV_STATUS_P16"]
HNSC_HPV_STATUS_CANDIDATES_PATIENT = ["HPV_STATUS", "SUBTYPE"]
HNSC_HPV_POS_TOKENS = {"hpv+", "positive", "pos", "hpv-positive"}
HNSC_HPV_NEG_TOKENS = {"hpv-", "negative", "neg", "hpv-negative"}
# ---------------------------------------------------------------------------
# GSE65858 — INDEPENDENT HPV validation cohort (GEO series). NAMED-SIDE
# ONLY — this dataset lives in `validate/` and is used exclusively for the
# reveal-side transfer test. It is NEVER anonymised, gets NO sealed map,
# and is NEVER passed to `engine`/`engine_v2`. The engine still discovers
# blind on TCGA; GSE65858 sees only the winner's already-revealed symbols.
# ---------------------------------------------------------------------------
GSE65858_GEO_ID = "GSE65858"
GSE65858_SERIES_MATRIX_URL = (
"https://ftp.ncbi.nlm.nih.gov/geo/series/"
"GSE65nnn/GSE65858/matrix/GSE65858_series_matrix.txt.gz"
)
# Illumina HumanHT-12 v4 platform — probe → HUGO Symbol annotation. If the
# NCBI FTP fetch fails, the build script prints instructions for a manual
# download and continues to look for the file in ``GSE65858_RAW_DIR``.
GSE65858_PLATFORM_ID = "GPL10558"
GSE65858_PLATFORM_URL = (
"https://ftp.ncbi.nlm.nih.gov/geo/platforms/"
"GPL10nnn/GPL10558/annot/GPL10558.annot.gz"
)
GSE65858_GEO_URL = (
f"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc={GSE65858_GEO_ID}"
)
GSE65858_RAW_DIR = REPO_ROOT / "data" / "raw_gse65858"
GSE65858_PROCESSED_DIR = REPO_ROOT / "data" / "processed_gse65858"
# Files the build expects in RAW_DIR.
GSE65858_FILES = {
"series_matrix": (
f"{GSE65858_GEO_ID}_series_matrix.txt.gz",
True,
),
"platform_annot": (
f"{GSE65858_PLATFORM_ID}.annot.gz",
# Optional — the build looks for a bundled probe→symbol map first,
# then falls back to parsing the platform annotation.
False,
),
}
# HPV+ definition (strict "virus-transcriptionally-active"). Anything
# else — DNA+/RNA-, DNA-, or unknown — reads as HPV-. The build derives
# these from the GSE65858 sample characteristics; the exact
# characteristic-field names live inside build_gse65858.py so schema.py
# stays lean.
GSE65858_HPV_POS_LABEL = "HPV+"
GSE65858_HPV_NEG_LABEL = "HPV-"
|