Download scripts/acquire_data.py from Hibbaan/Oncoct_v1: direct link, hf CLI and curl.
- Browser
- Download file 1.79 kB
-
https://huggingface.co/Hibbaan/Oncoct_v1/resolve/main/scripts/acquire_data.py
- Command line
-
hf download hf://Hibbaan/Oncoct_v1/scripts/acquire_data.py
-
curl -L -o acquire_data.py https://huggingface.co/Hibbaan/Oncoct_v1/resolve/main/scripts/acquire_data.py
1.79 kB
| #!/usr/bin/env python3 | |
| """Acquire pointers/manifests, not patient data by default.""" | |
| from __future__ import annotations | |
| import argparse, json, subprocess | |
| from pathlib import Path | |
| import requests | |
| CATALOG = { | |
| "msd": "s3://msd-for-monai/", | |
| "idc": "https://datacommons.cancer.gov/repository/imaging-data-commons", | |
| "lung_pet_ct_dx_manifest": "https://www.cancerimagingarchive.net/wp-content/uploads/Lung-PET-CT-Dx-NBIA-Manifest-122220.tcia", | |
| "hcc_tace_manifest": "https://www.cancerimagingarchive.net/wp-content/uploads/HCC-TACE-Seg_v1_202201.tcia", | |
| "nsclc_radiomics_manifest": "https://www.cancerimagingarchive.net/wp-content/uploads/NSCLC-Radiomics-Version-4-Oct-2020-NBIA-manifest.tcia", | |
| } | |
| def main(): | |
| p = argparse.ArgumentParser() | |
| p.add_argument("--dataset", choices=CATALOG, required=True) | |
| p.add_argument("--out", default="data_sources") | |
| p.add_argument("--download-manifest", action="store_true") | |
| a = p.parse_args(); out = Path(a.out); out.mkdir(parents=True, exist_ok=True) | |
| if a.dataset == "msd": | |
| subprocess.run(["aws", "s3", "ls", "s3://msd-for-monai/", "--no-sign-request"], check=True) | |
| print("Use aws s3 cp --no-sign-request for a selected MSD task; do not mirror all tasks unnecessarily.") | |
| elif a.dataset == "idc": | |
| (out / "idc_source.json").write_text(json.dumps({"source": CATALOG[a.dataset], "next": "Use idc-index or IDC BigQuery to generate a cohort manifest with license fields."}, indent=2)) | |
| else: | |
| target = out / (a.dataset + ".tcia") | |
| if a.download_manifest: | |
| r = requests.get(CATALOG[a.dataset], timeout=60); r.raise_for_status(); target.write_bytes(r.content) | |
| print(target) | |
| else: | |
| print(CATALOG[a.dataset]) | |
| if __name__ == "__main__": main() | |