chromatography-rt-prediction / revision /scripts /generate_revision_tables.py
AI4deeperScience's picture
Add files using upload-large-folder tool
6cf9dac verified
Raw
History Blame Contribute Delete
23.4 kB
"""Generate LaTeX macros and tables directly from frozen revision artifacts."""
from __future__ import annotations
import argparse
import hashlib
import json
from pathlib import Path
from typing import Any
import numpy as np
import pandas as pd
STRATEGIES = ("canonical_grouped", "scaffold_aware")
SEEDS = (123456, 123457, 123458)
STRATEGY_LABELS = {
"canonical_grouped": "Identity-grouped",
"scaffold_aware": "Scaffold-aware",
}
STRATEGY_MACROS = {
"canonical_grouped": "Canonical",
"scaffold_aware": "Scaffold",
}
MODEL_LABELS = {
"gat": "GAT",
"gcn": "GCN",
"fpnn": "FPNN",
"arithmetic_mean": "Arithmetic mean",
"stack_all_base_only": "Three-base stack",
"stack_all_plus_descriptors": "Three-base stack + descriptors",
"stack_gat_gcn": "GAT + GCN stack",
"stack_gat_fpnn": "GAT + FPNN stack",
"stack_gcn_fpnn": "GCN + FPNN stack",
"lab_median": "Laboratory median",
"descriptor_only_ridge": "Descriptors + laboratory ridge",
"descriptor_only_et": "Descriptors + laboratory ExtraTrees",
"fingerprint_no_lab_et": "Morgan ExtraTrees, no laboratory",
"fingerprint_no_lab_plus_lab_affine": "Morgan ExtraTrees + laboratory affine",
"fingerprint_only_et": "Morgan + one-hot laboratory ExtraTrees",
"fingerprint_plus_descriptors_et": "Morgan + descriptors + one-hot laboratory ExtraTrees",
}
def _load_json(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8"))
def _escape_latex(value: object) -> str:
text = str(value)
replacements = {
"\\": r"\textbackslash{}",
"&": r"\&",
"%": r"\%",
"$": r"\$",
"#": r"\#",
"_": r"\_",
"{": r"\{",
"}": r"\}",
}
return "".join(replacements.get(character, character) for character in text)
def _number(value: object, digits: int = 3, dash_for_missing: bool = True) -> str:
try:
numeric = float(value)
except (TypeError, ValueError):
return "--" if dash_for_missing else str(value)
if not np.isfinite(numeric):
return "--"
return f"{numeric:.{digits}f}"
def _mean_sd(row: pd.Series, metric: str) -> str:
return f"{_number(row[f'{metric}_mean'])} $\\pm$ {_number(row[f'{metric}_sd'])}"
def _write_metric_macros(
neural: pd.DataFrame,
inference_benchmark: dict[str, Any],
output_path: Path,
) -> None:
model_macros = {
"gat": "GAT",
"gcn": "GCN",
"fpnn": "FPNN",
"arithmetic_mean": "ArithmeticMean",
"stack_all_base_only": "BaseOnlyStack",
"stack_all_plus_descriptors": "Stack",
}
metric_macros = {"r2": "Rtwo", "mae": "MAE", "rmse": "RMSE"}
lines = ["% Generated from summary/neural_metrics_aggregate.csv."]
for strategy in STRATEGIES:
for model, model_macro in model_macros.items():
row = neural.loc[
(neural["strategy"] == strategy) & (neural["model"] == model)
]
if len(row) != 1:
raise ValueError(f"Expected one aggregate row for {strategy}/{model}.")
record = row.iloc[0]
for metric, metric_macro in metric_macros.items():
prefix = f"{STRATEGY_MACROS[strategy]}{model_macro}{metric_macro}"
lines.append(
f"\\newcommand{{\\{prefix}Mean}}{{{_number(record[f'{metric}_mean'])}}}"
)
lines.append(
f"\\newcommand{{\\{prefix}SD}}{{{_number(record[f'{metric}_sd'])}}}"
)
lines.extend(
[
f"\\newcommand{{\\InferenceBenchmarkN}}{{{int(inference_benchmark['n_records'])}}}",
f"\\newcommand{{\\InferenceBenchmarkFolds}}{{{int(inference_benchmark['n_fold_models_per_base'])}}}",
f"\\newcommand{{\\FeatureGenerationSeconds}}{{{_number(inference_benchmark['feature_generation']['seconds'])}}}",
f"\\newcommand{{\\FeaturePerRecordMilliseconds}}{{{_number(inference_benchmark['feature_generation']['milliseconds_per_record'])}}}",
f"\\newcommand{{\\InferenceBundleSeconds}}{{{_number(inference_benchmark['full_bundle']['seconds_mean'])}}}",
f"\\newcommand{{\\InferenceBundleSecondsSD}}{{{_number(inference_benchmark['full_bundle']['seconds_sd'])}}}",
f"\\newcommand{{\\InferencePerRecordMilliseconds}}{{{_number(inference_benchmark['full_bundle']['milliseconds_per_record_mean'])}}}",
]
)
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_dataset_table(audit: dict[str, Any], output_path: Path) -> None:
rows = (
("Structure--laboratory RT observations", audit["n_rows"]),
("Laboratories", audit["n_laboratories"]),
("Unique RDKit-canonical SMILES", audit["unique_canonical_isomeric_smiles"]),
("Unique full InChIKeys", audit["unique_full_inchi_keys"]),
("Unique InChIKey connectivity blocks", audit["unique_connectivity_blocks"]),
("Connectivity groups observed in multiple laboratories", audit["structures_seen_in_multiple_labs"]),
("Rows belonging to multi-laboratory connectivity groups", audit["rows_in_multi_lab_structures"]),
("Graph-conversion failures in supplied table", audit["graph_conversion_failures_in_provided_csv"]),
("Exact duplicate extra rows", audit["exact_duplicate_extra_rows"]),
("Structure--laboratory duplicate extra rows", audit["structure_lab_duplicate_extra_rows"]),
("Structure--laboratory groups with conflicting RT", audit["structure_lab_groups_with_conflicting_rt"]),
)
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Audited observational units and molecular-identity counts.}",
r"\label{tab:dataset-audit}",
r"\begin{tabular}{lr}",
r"\toprule",
r"Quantity & Count \\",
r"\midrule",
]
lines.extend(f"{_escape_latex(label)} & {int(value):,} \\\\" for label, value in rows)
lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _laboratory_ids(laboratories: list[str]) -> dict[str, str]:
return {laboratory: f"L{index:02d}" for index, laboratory in enumerate(laboratories, start=1)}
def _write_laboratory_counts(audit: dict[str, Any], output_path: Path) -> None:
counts = {str(key): int(value) for key, value in audit["laboratory_counts"].items()}
laboratories = sorted(counts)
ids = _laboratory_ids(laboratories)
lines = [
r"\begin{longtable}{llr}",
r"\caption{Laboratory identifiers and numbers of structure--laboratory observations.}\label{tab:lab-counts}\\",
r"\toprule",
r"ID & Laboratory & Observations \\",
r"\midrule",
r"\endfirsthead",
r"\toprule",
r"ID & Laboratory & Observations \\",
r"\midrule",
r"\endhead",
]
for laboratory in laboratories:
lines.append(
f"{ids[laboratory]} & {_escape_latex(laboratory)} & {counts[laboratory]:,} \\\\"
)
lines.extend([r"\bottomrule", r"\end{longtable}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_shared_lab_matrix(matrix_path: Path, output_path: Path) -> None:
matrix = pd.read_csv(matrix_path, index_col=0)
matrix.index = matrix.index.astype(str)
matrix.columns = matrix.columns.astype(str)
laboratories = sorted(set(matrix.index) | set(matrix.columns))
matrix = matrix.reindex(index=laboratories, columns=laboratories)
if matrix.isna().any().any():
raise ValueError("Shared-compound matrix is not square over the laboratory labels.")
ids = _laboratory_ids(laboratories)
column_specification = "l" + "r" * len(laboratories)
header = " & ".join(["ID"] + [ids[laboratory] for laboratory in laboratories]) + r" \\"
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Numbers of shared InChIKey-connectivity groups between every laboratory pair. Laboratory IDs are defined in Table~\ref{tab:lab-counts}; diagonal entries are the numbers of unique connectivity groups within each laboratory.}",
r"\label{tab:shared-lab-matrix}",
r"\resizebox{\textwidth}{!}{%",
f"\\begin{{tabular}}{{{column_specification}}}",
r"\toprule",
header,
r"\midrule",
]
for laboratory in laboratories:
values = " & ".join(str(int(matrix.loc[laboratory, other])) for other in laboratories)
lines.append(f"{ids[laboratory]} & {values} \\\\")
lines.extend(
[
r"\bottomrule",
r"\end{tabular}%",
r"}",
r"\end{table}",
]
)
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_split_table(artifacts_root: Path, output_path: Path) -> None:
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Outer split sizes and leakage checks for all predeclared runs.}",
r"\label{tab:split-audit}",
r"\begin{tabular}{lrrrrrr}",
r"\toprule",
r"Task & Seed & Development & Test & Test (\%) & Identity overlap & Laboratories \\",
r"\midrule",
]
for strategy in STRATEGIES:
for seed in SEEDS:
payload = _load_json(
artifacts_root / strategy / f"seed_{seed}" / "split_summary.json"
)
laboratories = (
f"{payload['development_laboratories']}/{payload['test_laboratories']}"
)
lines.append(
f"{STRATEGY_LABELS[strategy]} & {seed} & "
f"{payload['n_development_rows']:,} & {payload['n_test_rows']:,} & "
f"{100 * payload['test_fraction']:.1f} & "
f"{payload['structure_identity_overlap']} & {laboratories} \\\\"
)
lines.extend(
[
r"\bottomrule",
r"\end{tabular}",
r"\end{table}",
]
)
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_performance_table(neural: pd.DataFrame, output_path: Path) -> None:
model_order = ("gat", "gcn", "fpnn", "stack_all_plus_descriptors")
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Outer-test performance across three predeclared seeds (mean $\pm$ SD).}",
r"\label{tab:primary-performance}",
r"\begin{tabular}{llccc}",
r"\toprule",
r"Task & Model & $R^2$ & MAE (min) & RMSE (min) \\",
r"\midrule",
]
for strategy in STRATEGIES:
for model in model_order:
row = neural.loc[
(neural["strategy"] == strategy) & (neural["model"] == model)
].iloc[0]
lines.append(
f"{STRATEGY_LABELS[strategy]} & {MODEL_LABELS[model]} & "
f"{_mean_sd(row, 'r2')} & {_mean_sd(row, 'mae')} & "
f"{_mean_sd(row, 'rmse')} \\\\"
)
if strategy != STRATEGIES[-1]:
lines.append(r"\addlinespace")
lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_ablation_table(
neural: pd.DataFrame, classical: pd.DataFrame, output_path: Path
) -> None:
model_order = (
"arithmetic_mean",
"stack_all_base_only",
"stack_all_plus_descriptors",
"stack_gat_gcn",
"stack_gat_fpnn",
"stack_gcn_fpnn",
"lab_median",
"descriptor_only_ridge",
"fingerprint_no_lab_et",
"fingerprint_no_lab_plus_lab_affine",
"fingerprint_only_et",
"fingerprint_plus_descriptors_et",
)
source = pd.concat([neural, classical], ignore_index=True)
lines = [
r"\begin{longtable}{llcc}",
r"\caption{Fixed component ablations and classical baselines (mean $\pm$ SD across seeds).}\label{tab:ablations}\\",
r"\toprule",
r"Task & Model & $R^2$ & MAE (min) \\",
r"\midrule",
r"\endfirsthead",
r"\toprule",
r"Task & Model & $R^2$ & MAE (min) \\",
r"\midrule",
r"\endhead",
]
for strategy in STRATEGIES:
for model in model_order:
row = source.loc[
(source["strategy"] == strategy) & (source["model"] == model)
]
if row.empty:
continue
record = row.iloc[0]
lines.append(
f"{STRATEGY_LABELS[strategy]} & {_escape_latex(MODEL_LABELS[model])} & "
f"{_mean_sd(record, 'r2')} & {_mean_sd(record, 'mae')} \\\\"
)
if strategy != STRATEGIES[-1]:
lines.append(r"\addlinespace")
lines.extend([r"\bottomrule", r"\end{longtable}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_per_lab_table(per_lab: pd.DataFrame, output_path: Path) -> None:
primary = per_lab.loc[per_lab["model"] == "stack_all_plus_descriptors"].copy()
grouped = primary.groupby(["strategy", "Lab"], sort=True)
lines = [
r"\begin{landscape}",
r"\begingroup",
r"\scriptsize",
r"\renewcommand{\arraystretch}{0.92}",
r"\setlength{\tabcolsep}{3pt}",
r"\begin{longtable}{p{1.15in}p{2.2in}rccccc}",
r"\caption{Per-laboratory full-stack metrics across outer seeds.}\label{tab:per-lab}\\",
r"\toprule",
r"Task & Laboratory & $n$ range & MAE & RMSE & Bias & Slope & nMAE \\",
r"\midrule",
r"\endfirsthead",
r"\toprule",
r"Task & Laboratory & $n$ range & MAE & RMSE & Bias & Slope & nMAE \\",
r"\midrule",
r"\endhead",
]
for (strategy, laboratory), frame in grouped:
n_range = f"{int(frame['n'].min())}--{int(frame['n'].max())}"
values = []
for column in ("mae", "rmse", "bias", "calibration_slope", "nmae_by_rt_range"):
mean = frame[column].mean()
sd = frame[column].std(ddof=1)
values.append(f"{_number(mean)} $\\pm$ {_number(sd)}")
lines.append(
f"{STRATEGY_LABELS[strategy]} & {_escape_latex(laboratory)} & {n_range} & "
+ " & ".join(values)
+ r" \\"
)
lines.extend(
[
r"\bottomrule",
r"\end{longtable}",
r"\endgroup",
r"\end{landscape}",
]
)
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_domain_table(domain: pd.DataFrame, output_path: Path) -> None:
grouped = domain.groupby(["strategy", "threshold"], sort=True)
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Prospective maximum-Tanimoto threshold sensitivity across outer seeds.}",
r"\label{tab:domain}",
r"\begin{tabular}{lccccc}",
r"\toprule",
r"Task & Threshold & Coverage & Accepted MAE & Rejected MAE & $\rho$(similarity, $|e|$) \\",
r"\midrule",
]
for (strategy, threshold), frame in grouped:
lines.append(
f"{STRATEGY_LABELS[strategy]} & {threshold:.1f} & "
f"{_number(frame['accepted_coverage'].mean())} $\\pm$ {_number(frame['accepted_coverage'].std(ddof=1))} & "
f"{_number(frame['accepted_mae'].mean())} $\\pm$ {_number(frame['accepted_mae'].std(ddof=1))} & "
f"{_number(frame['rejected_mae'].mean())} $\\pm$ {_number(frame['rejected_mae'].std(ddof=1))} & "
f"{_number(frame['spearman_similarity_vs_absolute_error'].mean())} \\\\"
)
lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_paired_table(paired: pd.DataFrame, output_path: Path) -> None:
selected = paired.loc[
(paired["reference_model"].isin(["gat", "gcn", "fpnn"]))
& (paired["metric"] == "mae")
].copy()
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Paired connectivity-group bootstrap differences in MAE for the full stack relative to each base model. Negative differences favor the full stack.}",
r"\label{tab:paired-bootstrap}",
r"\begin{tabular}{llrr}",
r"\toprule",
r"Task & Reference & Seed & Difference (95\% interval), min \\",
r"\midrule",
]
for _, row in selected.sort_values(["strategy", "reference_model", "seed"]).iterrows():
lines.append(
f"{STRATEGY_LABELS[row['strategy']]} & {MODEL_LABELS[row['reference_model']]} & "
f"{int(row['seed'])} & {_number(row['difference_point'])} "
f"[{_number(row['ci_low'])}, {_number(row['ci_high'])}] \\\\"
)
lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_embedding_stability_table(
embedding_stability: pd.DataFrame, output_path: Path
) -> None:
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Cross-seed stability of learned laboratory geometry. Values are Spearman correlations between vectors of mean foldwise pairwise laboratory distances; the interval gives the minimum and maximum across the three seed pairs.}",
r"\label{tab:embedding-stability}",
r"\begin{tabular}{llc}",
r"\toprule",
r"Task & Model & Spearman $\rho$ (range) \\",
r"\midrule",
]
for strategy in STRATEGIES:
for model in ("gat", "gcn", "fpnn"):
row = embedding_stability.loc[
(embedding_stability["strategy"] == strategy)
& (embedding_stability["model"] == model)
]
if len(row) != 1:
raise ValueError(
f"Expected one embedding-stability row for {strategy}/{model}."
)
record = row.iloc[0]
lines.append(
f"{STRATEGY_LABELS[strategy]} & {MODEL_LABELS[model]} & "
f"{_number(record['mean'])} "
f"[{_number(record['min'])}, {_number(record['max'])}] \\\\"
)
if strategy != STRATEGIES[-1]:
lines.append(r"\addlinespace")
lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_computational_cost_table(runtime: pd.DataFrame, output_path: Path) -> None:
lines = [
r"\begin{table}[htbp]",
r"\centering",
r"\caption{Trainable parameter counts and cumulative six-fold training time per outer run. Times are mean $\pm$ SD across three seeds on the hardware reported in the text.}",
r"\label{tab:computational-cost}",
r"\begin{tabular}{llrr}",
r"\toprule",
r"Task & Model & Parameters & Training time (min) \\",
r"\midrule",
]
for strategy in STRATEGIES:
for model in ("gat", "gcn", "fpnn"):
row = runtime.loc[
(runtime["strategy"] == strategy) & (runtime["model"] == model)
]
if len(row) != 1:
raise ValueError(f"Expected one runtime row for {strategy}/{model}.")
record = row.iloc[0]
mean_minutes = float(record["training_seconds_mean"]) / 60.0
sd_minutes = float(record["training_seconds_sd"]) / 60.0
lines.append(
f"{STRATEGY_LABELS[strategy]} & {MODEL_LABELS[model]} & "
f"{int(record['parameter_count']):,} & "
f"{mean_minutes:.2f} $\\pm$ {sd_minutes:.2f} \\\\"
)
if strategy != STRATEGIES[-1]:
lines.append(r"\addlinespace")
lines.extend([r"\bottomrule", r"\end{tabular}", r"\end{table}"])
output_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
def _write_table_manifest(output_dir: Path) -> None:
rows = [
{
"relative_path": path.name,
"bytes": path.stat().st_size,
"sha256": hashlib.sha256(path.read_bytes()).hexdigest(),
}
for path in sorted(output_dir.glob("*.tex"))
]
pd.DataFrame(rows).to_csv(output_dir / "TABLE_MANIFEST.csv", index=False)
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--artifacts-root", required=True)
parser.add_argument("--output-dir", required=True)
args = parser.parse_args()
artifacts_root = Path(args.artifacts_root).resolve()
output_dir = Path(args.output_dir).resolve()
if output_dir.exists() and any(output_dir.iterdir()):
raise FileExistsError(f"Refusing nonempty generated-table directory: {output_dir}")
output_dir.mkdir(parents=True, exist_ok=True)
summary_dir = artifacts_root / "summary"
neural = pd.read_csv(summary_dir / "neural_metrics_aggregate.csv")
classical = pd.read_csv(summary_dir / "classical_metrics_aggregate.csv")
per_lab = pd.read_csv(summary_dir / "neural_per_lab_by_run.csv")
domain = pd.read_csv(summary_dir / "prospective_domain_by_run.csv")
paired = pd.read_csv(summary_dir / "paired_bootstrap_by_run.csv")
embedding_stability = pd.read_csv(
summary_dir
/ "embedding_stability"
/ "embedding_stability_aggregate.csv"
)
runtime = pd.read_csv(summary_dir / "neural_runtime_aggregate.csv")
audit = _load_json(artifacts_root / "dataset_audit.json")
inference_benchmark = _load_json(artifacts_root / "inference_benchmark.json")
_write_metric_macros(
neural,
inference_benchmark,
output_dir / "results_macros.tex",
)
_write_dataset_table(audit, output_dir / "table_dataset_audit.tex")
_write_laboratory_counts(audit, output_dir / "table_laboratory_counts.tex")
_write_shared_lab_matrix(
artifacts_root / "shared_compounds_by_lab.csv",
output_dir / "table_shared_lab_matrix.tex",
)
_write_split_table(artifacts_root, output_dir / "table_split_audit.tex")
_write_performance_table(neural, output_dir / "table_performance.tex")
_write_ablation_table(neural, classical, output_dir / "table_ablations.tex")
_write_per_lab_table(per_lab, output_dir / "table_per_lab.tex")
_write_domain_table(domain, output_dir / "table_domain.tex")
_write_paired_table(paired, output_dir / "table_paired_bootstrap.tex")
_write_embedding_stability_table(
embedding_stability,
output_dir / "table_embedding_stability.tex",
)
_write_computational_cost_table(
runtime,
output_dir / "table_computational_cost.tex",
)
_write_table_manifest(output_dir)
print(f"Wrote revision macros and LaTeX tables to: {output_dir}")
return 0
if __name__ == "__main__":
raise SystemExit(main())