chromatography-rt-prediction / revision /scripts /summarize_reanalysis.py
AI4deeperScience's picture
Add files using upload-large-folder tool
8f4ed7a verified
Raw
History Blame Contribute Delete
7.62 kB
"""Aggregate the six frozen reviewer-requested outer runs."""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from typing import Any
import pandas as pd
PROJECT_ROOT = Path(__file__).resolve().parents[2]
if str(PROJECT_ROOT) not in sys.path:
sys.path.insert(0, str(PROJECT_ROOT))
from revision.scripts.reanalysis_core import ensure_new_output_dir
from revision.scripts.reanalysis_pipeline import _write_sha256_manifest
STRATEGIES = ("canonical_grouped", "scaffold_aware")
SEEDS = (123456, 123457, 123458)
METRIC_COLUMNS = (
"r2",
"mae",
"rmse",
"bias",
"calibration_slope",
"calibration_intercept",
)
def _load_json(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8"))
def _metric_rows(
strategy: str,
seed: int,
payload: dict[str, Any],
) -> list[dict[str, Any]]:
return [
{
"strategy": strategy,
"seed": seed,
"primary_model_predeclared": payload["primary_model_predeclared"],
"model": model,
**metrics,
}
for model, metrics in payload["metrics"].items()
]
def _aggregate_metrics(frame: pd.DataFrame) -> pd.DataFrame:
metric_columns = [column for column in METRIC_COLUMNS if column in frame.columns]
aggregate = frame.groupby(["strategy", "model"])[metric_columns].agg(["mean", "std"])
aggregate.columns = [
f"{metric}_{'sd' if statistic == 'std' else statistic}"
for metric, statistic in aggregate.columns
]
return aggregate.reset_index()
def _aggregate_runtime(frame: pd.DataFrame) -> pd.DataFrame:
aggregate = (
frame.groupby(["strategy", "model"], sort=True)
.agg(
parameter_count=("parameter_count", "first"),
parameter_count_min=("parameter_count", "min"),
parameter_count_max=("parameter_count", "max"),
training_seconds_mean=("training_seconds", "mean"),
training_seconds_sd=("training_seconds", "std"),
)
.reset_index()
)
if not (
aggregate["parameter_count"] == aggregate["parameter_count_min"]
).all() or not (
aggregate["parameter_count"] == aggregate["parameter_count_max"]
).all():
raise ValueError("Parameter counts changed across frozen outer runs.")
return aggregate.drop(columns=["parameter_count_min", "parameter_count_max"])
def main() -> int:
parser = argparse.ArgumentParser(
description="Aggregate all six completed frozen reviewer reanalysis runs."
)
parser.add_argument(
"--artifacts-root",
default=str(
PROJECT_ROOT
/ "revision"
/ "artifacts"
/ "reviewer_requested_reanalysis_v4"
),
)
artifacts_root = Path(parser.parse_args().artifacts_root).resolve()
summary_dir = ensure_new_output_dir(artifacts_root / "summary")
classical_rows: list[dict[str, Any]] = []
neural_rows: list[dict[str, Any]] = []
paired_rows: list[dict[str, Any]] = []
domain_rows: list[dict[str, Any]] = []
per_lab_frames: list[pd.DataFrame] = []
runtime_rows: list[dict[str, Any]] = []
for strategy in STRATEGIES:
for seed in SEEDS:
split_dir = artifacts_root / strategy / f"seed_{seed}"
neural_dir = split_dir / "neural_stack"
neural_metrics_path = neural_dir / "metrics.json"
if not neural_metrics_path.is_file():
raise FileNotFoundError(f"Incomplete neural matrix: {neural_metrics_path}")
classical_rows.extend(
_metric_rows(strategy, seed, _load_json(split_dir / "classical_metrics.json"))
)
neural_rows.extend(
_metric_rows(strategy, seed, _load_json(neural_metrics_path))
)
for reference_model, metric_payload in _load_json(
neural_dir / "paired_group_bootstrap.json"
).items():
for metric, values in metric_payload.items():
paired_rows.append(
{
"strategy": strategy,
"seed": seed,
"candidate_model": "stack_all_plus_descriptors",
"reference_model": reference_model,
"metric": metric,
**values,
}
)
domain = _load_json(neural_dir / "prospective_domain_diagnostics.json")
for threshold_payload in domain["threshold_sensitivity"]:
domain_rows.append(
{
"strategy": strategy,
"seed": seed,
"spearman_similarity_vs_absolute_error": domain[
"spearman_similarity_vs_absolute_error"
],
"spearman_model_spread_vs_absolute_error": domain[
"spearman_model_spread_vs_absolute_error"
],
**threshold_payload,
}
)
per_lab = pd.read_csv(neural_dir / "per_lab_metrics.csv")
per_lab.insert(0, "seed", seed)
per_lab.insert(0, "strategy", strategy)
per_lab_frames.append(per_lab)
runtime = _load_json(neural_dir / "RUN_METADATA.json")
for model, training_seconds in runtime["training_seconds_by_model"].items():
runtime_rows.append(
{
"strategy": strategy,
"seed": seed,
"model": model,
"parameter_count": int(runtime["parameter_counts"][model]),
"training_seconds": float(training_seconds),
"total_run_seconds": float(runtime["total_run_seconds"]),
"device": runtime["device"],
"gpu_name": runtime.get("gpu_name"),
}
)
classical = pd.DataFrame(classical_rows)
neural = pd.DataFrame(neural_rows)
classical.to_csv(summary_dir / "classical_metrics_by_run.csv", index=False)
neural.to_csv(summary_dir / "neural_metrics_by_run.csv", index=False)
_aggregate_metrics(classical).to_csv(
summary_dir / "classical_metrics_aggregate.csv", index=False
)
_aggregate_metrics(neural).to_csv(
summary_dir / "neural_metrics_aggregate.csv", index=False
)
pd.DataFrame(paired_rows).to_csv(
summary_dir / "paired_bootstrap_by_run.csv", index=False
)
pd.DataFrame(domain_rows).to_csv(
summary_dir / "prospective_domain_by_run.csv", index=False
)
pd.concat(per_lab_frames, ignore_index=True).to_csv(
summary_dir / "neural_per_lab_by_run.csv", index=False
)
runtime_frame = pd.DataFrame(runtime_rows)
runtime_frame.to_csv(summary_dir / "neural_runtime_by_run.csv", index=False)
_aggregate_runtime(runtime_frame).to_csv(
summary_dir / "neural_runtime_aggregate.csv", index=False
)
_write_sha256_manifest(artifacts_root)
print(f"Wrote frozen aggregate tables to: {summary_dir}")
return 0
if __name__ == "__main__":
raise SystemExit(main())