File size: 4,774 Bytes
ef53368
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fe82c54
ef53368
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fe82c54
 
 
ef53368
 
 
 
 
 
 
 
 
 
 
 
 
 
 
fe82c54
 
 
 
 
 
 
 
ef53368
fe82c54
ef53368
 
 
 
fe82c54
 
 
ef53368
 
 
 
90b47cf
 
 
ef53368
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
"""Source of truth for the Tables S10/S11 notebook (MagNET-Zero/MagNET-PCM scaling factors). Edit
the cell sources here, then regenerate:

    python3 build_nb_scaling.py si_table_s10_s11_scaling
    jupyter nbconvert --to notebook --execute --inplace ../si_tables/si_table_s10_s11_scaling.ipynb

Regenerating overwrites the .ipynb (clearing its execution outputs). All real code lives in
scaling_factors.py (the published tables and the re-derivation from delta-22); the notebook only
runs it. The notebook lives in analysis/si_tables/ (grouped by role), not here.

Run with no arguments to list the available notebook names.
"""
import os
import sys

from nb_build import md, code, save_notebook as _save


_BOOTSTRAP = r"""
import os, sys

# make the in-repo modules importable (not pip-installed)
REPO = os.path.abspath("../..")
for _p in ("data/delta22", "data/applications",
           "analysis/code", "analysis/code/shared"):
    sys.path.insert(0, os.path.join(REPO, _p))
"""

_IMPORTS = r"""
import numpy as np
import pandas as pd
import scaling_factors
import build_composite_model
import paths
from applications_reader import Applications
"""

_SETUP = r"""
DELTA22 = paths.dataset_file("delta22", root=REPO)
XLSX = os.path.join(REPO, "data", "delta22", "delta22_experimental.xlsx")

def document_path(name):
    os.makedirs("documents", exist_ok=True)
    return os.path.join("documents", name)
"""

si_table_s10_s11_scaling = [
    md(r"""
# Tables S10 and S11: MagNET-Zero/MagNET-PCM scaling factors

Per-solvent linear coefficients (intercept, stationary, pcm) mapping MagNET-Zero shieldings plus a
MagNET-PCM correction to predicted shifts: ¹H (S10) and ¹³C (S11). The published coefficients are
reflection-symmetrized (each MagNET-Zero/PCM shielding averaged over twenty forward passes, the input
geometry mirrored for half), which removes the model's reflection-parity error.
"""),
    code(_BOOTSTRAP),
    code(_IMPORTS),
    code(_SETUP),
    code(r"""
tables = scaling_factors.published_scaling_tables()   # {"H": Table S10, "C": Table S11}
print("Table S10 (1H):"); display(tables["H"])
print("Table S11 (13C):"); display(tables["C"])

out = document_path("si_table_s10_s11_scaling.xlsx")
with pd.ExcelWriter(out) as writer:
    tables["H"].reset_index().to_excel(writer, sheet_name="Table S10 (1H)", index=False)
    tables["C"].reset_index().to_excel(writer, sheet_name="Table S11 (13C)", index=False)
print("wrote", os.path.relpath(out, REPO))
"""),
    md(r"""
## Reproducibility check: re-derive from delta-22

The published tables are reflection-symmetrized, so reproducing them exactly needs the model
checkpoints (`build_scaling_tables(symmetrized=True)`). The checkpoint-free
`build_scaling_tables(symmetrized=False)` below fits the raw single-pass shieldings stored in the
released delta-22 file and lands within ~0.01 ppm.
"""),
    code(r"""
derived = scaling_factors.build_scaling_tables(DELTA22, XLSX)   # symmetrized=False (checkpoint-free)
for nucleus in ("H", "C"):
    p = tables[nucleus]
    d = derived[nucleus].reindex(p.index)[p.columns]
    max_dev = float(np.abs(p.values - d.values).max())
    print(f"{nucleus}: largest published-vs-raw-refit deviation = {max_dev:.2e} ppm")
    assert max_dev < 1.5e-2, f"{nucleus} raw refit drifted too far from the published values"
print("raw delta-22 refit reproduces the published tables to the reflection-parity floor")
"""),
    md(r"""
## Composite-model coefficients (Figure 5C/5D, SI S15)

The natural-products figures use a larger family of per-solvent coefficients fit on delta-22: OLS
fits, 1000-seed bootstrap resamples, their RMSE distributions, and the PCM conversion factors. These
regenerate from released inputs alone.
"""),
    code(r"""
reader = Applications(paths.dataset_file("applications", root=REPO))
regenerated = build_composite_model.regenerate(reader)
max_dev = build_composite_model.verify(reader, regenerated)
print(f"composite_model largest regenerated-vs-stored deviation = {max_dev:.2e}")
"""),
]


# name -> (cells, path relative to repo root)
NOTEBOOKS = {
    "si_table_s10_s11_scaling": (si_table_s10_s11_scaling,
                                 "analysis/si_tables/si_table_s10_s11_scaling.ipynb"),
}

if __name__ == "__main__":
    here = os.path.dirname(os.path.abspath(__file__))
    repo = os.path.abspath(os.path.join(here, "..", ".."))
    names = sys.argv[1:]
    if not names:
        print("available names:", ", ".join(NOTEBOOKS))
        sys.exit(0)
    unknown = [n for n in names if n not in NOTEBOOKS]
    if unknown:
        raise SystemExit(f"unknown notebook name(s): {unknown}; available: {', '.join(NOTEBOOKS)}")
    for name in names:
        cells, relpath = NOTEBOOKS[name]
        _save(cells, os.path.join(repo, relpath))