File size: 6,237 Bytes
0fff343
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
"""Pinned constants for the cBioPortal CRC dataset we depend on.

If cBioPortal renames a file or column, this is the single place to update.
"""

from pathlib import Path

STUDY = "coadread_tcga_pan_can_atlas_2018"

# Datahub LFS-resolved URL. The S3 tarball mirror returns 403 and the plain
# raw.githubusercontent.com URL returns 131-byte LFS pointers; this is the URL
# pattern that yields real content.
BASE_URL = (
    "https://media.githubusercontent.com/media/cBioPortal/datahub/master/public/"
    + STUDY
)

# Human study page (for the manual-download fallback message).
STUDY_PAGE_URL = f"https://www.cbioportal.org/study/summary?id={STUDY}"
DATAHUB_BROWSE_URL = (
    f"https://github.com/cBioPortal/datahub/tree/master/public/{STUDY}"
)

# Logical name -> (filename, required_for_chunk1).
# `data_mutations.txt` is ~340 MB and currently over the datahub's GitHub LFS
# budget; we treat it as optional. Chunk 1 doesn't need it (TMB comes from the
# clinical sample file). Later chunks that want per-variant calls will need to
# fetch it from cBioPortal directly.
FILES = {
    "sample":     ("data_clinical_sample.txt",  True),
    "patient":    ("data_clinical_patient.txt", True),
    "expression": ("data_mrna_seq_v2_rsem.txt", True),
    "mutations":  ("data_mutations.txt",        False),
}

# Project paths.
REPO_ROOT = Path(__file__).resolve().parent.parent
RAW_DIR = REPO_ROOT / "data" / "raw"
PROCESSED_DIR = REPO_ROOT / "data" / "processed"

# Columns we depend on per file. build.py fails loudly if any are absent, listing
# the columns it *did* find.
REQUIRED_SAMPLE_COLS = [
    "PATIENT_ID",
    "SAMPLE_ID",
    "ONCOTREE_CODE",         # COAD vs READ
    "MSI_SENSOR_SCORE",      # continuous MSI score; thresholded into msi_status
    "TMB_NONSYNONYMOUS",
]
REQUIRED_PATIENT_COLS = [
    "PATIENT_ID",
    "AGE",
    "SEX",
    "AJCC_PATHOLOGIC_TUMOR_STAGE",
    "OS_STATUS",
    "OS_MONTHS",
]

# MSI label derivation. Thresholds are taken verbatim from the suggestion in the
# cBioPortal data_clinical_sample.txt header:
#   "MSI Score reported by MSIsensor. The suggested thresholds are
#    MSI: >10, Indeterminate: 4-10 and MSS: <10."
# We use >= for the boundary.
MSI_SENSOR_HIGH = 10.0   # >= this => MSI-H
MSI_SENSOR_LOW = 4.0     # < this  => MSS; in-between => Indeterminate


# ---------------------------------------------------------------------------
# HNSC (head & neck) — second dataset. Mirrors the CRC layout above.
# ---------------------------------------------------------------------------
#
# Study: hnsc_tcga_pan_can_atlas_2018 — TCGA HNSC PanCancer Atlas.
# HPV status field: depending on the cBioPortal release the called
# "HPV+/HPV-" lives EITHER in the sample clinical file (column
# ``HPV_STATUS``) or is derivable from the patient ``SUBTYPE`` (suffix
# ``_HPV+`` / ``_HPV-``). The build script tries both, in that order,
# and fails loudly if neither is present.

HNSC_STUDY = "hnsc_tcga_pan_can_atlas_2018"
HNSC_BASE_URL = (
    "https://media.githubusercontent.com/media/cBioPortal/datahub/master/public/"
    + HNSC_STUDY
)
HNSC_STUDY_PAGE_URL = f"https://www.cbioportal.org/study/summary?id={HNSC_STUDY}"
HNSC_DATAHUB_BROWSE_URL = (
    f"https://github.com/cBioPortal/datahub/tree/master/public/{HNSC_STUDY}"
)

HNSC_FILES = {
    "sample":     ("data_clinical_sample.txt",   True),
    "patient":    ("data_clinical_patient.txt",  True),
    "expression": ("data_mrna_seq_v2_rsem.txt",  True),
}

HNSC_RAW_DIR = REPO_ROOT / "data" / "raw_hnsc"
HNSC_PROCESSED_DIR = REPO_ROOT / "data" / "processed_hnsc"

# Columns we depend on. HPV_STATUS may be absent on some releases — see the
# fallback in build_hnsc.py.
HNSC_REQUIRED_SAMPLE_COLS = ["PATIENT_ID", "SAMPLE_ID"]
HNSC_REQUIRED_PATIENT_COLS = ["PATIENT_ID", "AGE", "SEX", "AJCC_PATHOLOGIC_TUMOR_STAGE"]

# Columns / value patterns the build script will look for, in priority order.
HNSC_HPV_STATUS_CANDIDATES_SAMPLE = ["HPV_STATUS", "HPV_STATUS_ISH", "HPV_STATUS_P16"]
HNSC_HPV_STATUS_CANDIDATES_PATIENT = ["HPV_STATUS", "SUBTYPE"]
HNSC_HPV_POS_TOKENS = {"hpv+", "positive", "pos", "hpv-positive"}
HNSC_HPV_NEG_TOKENS = {"hpv-", "negative", "neg", "hpv-negative"}


# ---------------------------------------------------------------------------
# GSE65858 — INDEPENDENT HPV validation cohort (GEO series). NAMED-SIDE
# ONLY — this dataset lives in `validate/` and is used exclusively for the
# reveal-side transfer test. It is NEVER anonymised, gets NO sealed map,
# and is NEVER passed to `engine`/`engine_v2`. The engine still discovers
# blind on TCGA; GSE65858 sees only the winner's already-revealed symbols.
# ---------------------------------------------------------------------------
GSE65858_GEO_ID = "GSE65858"
GSE65858_SERIES_MATRIX_URL = (
    "https://ftp.ncbi.nlm.nih.gov/geo/series/"
    "GSE65nnn/GSE65858/matrix/GSE65858_series_matrix.txt.gz"
)
# Illumina HumanHT-12 v4 platform — probe → HUGO Symbol annotation. If the
# NCBI FTP fetch fails, the build script prints instructions for a manual
# download and continues to look for the file in ``GSE65858_RAW_DIR``.
GSE65858_PLATFORM_ID = "GPL10558"
GSE65858_PLATFORM_URL = (
    "https://ftp.ncbi.nlm.nih.gov/geo/platforms/"
    "GPL10nnn/GPL10558/annot/GPL10558.annot.gz"
)
GSE65858_GEO_URL = (
    f"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc={GSE65858_GEO_ID}"
)

GSE65858_RAW_DIR = REPO_ROOT / "data" / "raw_gse65858"
GSE65858_PROCESSED_DIR = REPO_ROOT / "data" / "processed_gse65858"

# Files the build expects in RAW_DIR.
GSE65858_FILES = {
    "series_matrix": (
        f"{GSE65858_GEO_ID}_series_matrix.txt.gz",
        True,
    ),
    "platform_annot": (
        f"{GSE65858_PLATFORM_ID}.annot.gz",
        # Optional — the build looks for a bundled probe→symbol map first,
        # then falls back to parsing the platform annotation.
        False,
    ),
}

# HPV+ definition (strict "virus-transcriptionally-active"). Anything
# else — DNA+/RNA-, DNA-, or unknown — reads as HPV-. The build derives
# these from the GSE65858 sample characteristics; the exact
# characteristic-field names live inside build_gse65858.py so schema.py
# stays lean.
GSE65858_HPV_POS_LABEL = "HPV+"
GSE65858_HPV_NEG_LABEL = "HPV-"