anhtld commited on
Commit
d671906
·
verified ·
1 Parent(s): 4fafef9

manual-selector-sweep-paper-sync 2026-07-04T04:48:53Z workspace

Browse files
workspace/README.md CHANGED
@@ -292,6 +292,10 @@ Dominance and utility:
292
  features. Use `--no-markdown-report` for README-only runs.
293
  - `scripts/eval_nonlinear_dominance_selector.py`: nonlinear selector sweep.
294
  Use `--no-markdown-report` for README-only runs.
 
 
 
 
295
 
296
  Metric and paper artifacts:
297
 
@@ -443,6 +447,9 @@ High-value run directories:
443
  - `runs/ctt_base_context_obs_learned_dominance_*_perdim_trainmax_train_to_test`:
444
  per-dimension trainmax selector diagnostics from Slurm job `15149815`; these
445
  are below the current K=16 `env_clip` support setting.
 
 
 
446
  - `runs/ctt_base_context_obs_nonlinear_dominance_chartcompat_obs_*`: fixed
447
  nonlinear selector diagnostics.
448
  - `runs/summary_ctt.csv`: global run summary table.
 
292
  features. Use `--no-markdown-report` for README-only runs.
293
  - `scripts/eval_nonlinear_dominance_selector.py`: nonlinear selector sweep.
294
  Use `--no-markdown-report` for README-only runs.
295
+ - `scripts/build_selector_diagnostic_sweep.py`: non-cherry-picked selector
296
+ diagnostic summary builder. It reads completed selector `metrics.json`
297
+ artifacts, keeps all candidate rows, selects the best row per action
298
+ convention by held-out selected success, and writes JSON/TeX/log outputs.
299
 
300
  Metric and paper artifacts:
301
 
 
447
  - `runs/ctt_base_context_obs_learned_dominance_*_perdim_trainmax_train_to_test`:
448
  per-dimension trainmax selector diagnostics from Slurm job `15149815`; these
449
  are below the current K=16 `env_clip` support setting.
450
+ - `runs/ctt_selector_diagnostic_sweep`: generated selector summary table used
451
+ by the paper to compare K=8 tanh, K=8 per-dim trainmax, and K=16 env-clip
452
+ selector diagnostics without cherry-picking a single row.
453
  - `runs/ctt_base_context_obs_nonlinear_dominance_chartcompat_obs_*`: fixed
454
  nonlinear selector diagnostics.
455
  - `runs/summary_ctt.csv`: global run summary table.
workspace/latex/main.aux CHANGED
@@ -80,64 +80,66 @@
80
  \bibcite{singh2026bokbo}{7}
81
  \bibcite{tao2024maniskill3}{8}
82
  \bibcite{zhang2024vlabench}{9}
83
- \bibcite{zhao2026verispace}{10}
84
  \@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{15}{section.10}\protected@file@percent }
85
- \@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces No-clipping validation refresh for \texttt {base\_context\_obs}, K=8. This replays decoded raw actions from restored states and records action-bound validity labels before any clipping. The labels are action-space violations, not collision/contact outcomes.}}{16}{table.9}\protected@file@percent }
86
- \newlabel{tab:ctt-base-context-obs-val-noclip-rollout}{{9}{16}{No-clipping validation refresh for \texttt {base\_context\_obs}, K=8. This replays decoded raw actions from restored states and records action-bound validity labels before any clipping. The labels are action-space violations, not collision/contact outcomes}{table.9}{}}
87
- \@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces No-clipping held-out test refresh for \texttt {base\_context\_obs}, K=8. The proposal oracle remains nonzero, but selected success does not improve over the raw-replay base and almost all known labels indicate action-space bound violations.}}{17}{table.10}\protected@file@percent }
88
- \newlabel{tab:ctt-base-context-obs-test-noclip-rollout}{{10}{17}{No-clipping held-out test refresh for \texttt {base\_context\_obs}, K=8. The proposal oracle remains nonzero, but selected success does not improve over the raw-replay base and almost all known labels indicate action-space bound violations}{table.10}{}}
89
- \@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Scaled raw-action validation refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. This is a diagnostic convention suggested by the action-bound audit, not a final action representation.}}{18}{table.11}\protected@file@percent }
90
- \newlabel{tab:ctt-base-context-obs-val-scaled-rollout}{{11}{18}{Scaled raw-action validation refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. This is a diagnostic convention suggested by the action-bound audit, not a final action representation}{table.11}{}}
91
- \@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Scaled raw-action held-out test refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. The global scale improves base action-bound validity but does not preserve proposal support.}}{19}{table.12}\protected@file@percent }
92
- \newlabel{tab:ctt-base-context-obs-test-scaled-rollout}{{12}{19}{Scaled raw-action held-out test refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. The global scale improves base action-bound validity but does not preserve proposal support}{table.12}{}}
93
- \@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Per-dimension train-max scaled validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The scale vector is fit only from the train split action-bound audit.}}{20}{table.13}\protected@file@percent }
94
- \newlabel{tab:ctt-base-context-obs-val-perdim-rollout}{{13}{20}{Per-dimension train-max scaled validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The scale vector is fit only from the train split action-bound audit}{table.13}{}}
95
- \@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Per-dimension train-max scaled held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This diagnostic tests whether per-dimension max-fit scaling preserves more support than the global scale.}}{21}{table.14}\protected@file@percent }
96
- \newlabel{tab:ctt-base-context-obs-test-perdim-rollout}{{14}{21}{Per-dimension train-max scaled held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This diagnostic tests whether per-dimension max-fit scaling preserves more support than the global scale}{table.14}{}}
97
- \@writefile{lot}{\contentsline {table}{\numberline {15}{\ignorespaces Explicit \texttt {env\_clip} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The declared decoder convention clips decoded controls to action-space bounds before validity checks, so action-bound labels measure the declared convention rather than silent simulator clipping.}}{22}{table.15}\protected@file@percent }
98
- \newlabel{tab:ctt-base-context-obs-val-envclip-rollout}{{15}{22}{Explicit \texttt {env\_clip} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The declared decoder convention clips decoded controls to action-space bounds before validity checks, so action-bound labels measure the declared convention rather than silent simulator clipping}{table.15}{}}
99
- \@writefile{lot}{\contentsline {table}{\numberline {16}{\ignorespaces Explicit \texttt {env\_clip} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This convention is action-bound-clean and preserves more proposal support than bounded \texttt {tanh}, but score-only selection remains below base.}}{23}{table.16}\protected@file@percent }
100
- \newlabel{tab:ctt-base-context-obs-test-envclip-rollout}{{16}{23}{Explicit \texttt {env\_clip} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This convention is action-bound-clean and preserves more proposal support than bounded \texttt {tanh}, but score-only selection remains below base}{table.16}{}}
101
- \@writefile{lot}{\contentsline {table}{\numberline {17}{\ignorespaces Bounded \texttt {tanh} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This action convention maps decoded controls into finite action bounds before validity checks.}}{24}{table.17}\protected@file@percent }
102
- \newlabel{tab:ctt-base-context-obs-val-tanh-rollout}{{17}{24}{Bounded \texttt {tanh} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This action convention maps decoded controls into finite action bounds before validity checks}{table.17}{}}
103
- \@writefile{lot}{\contentsline {table}{\numberline {18}{\ignorespaces Bounded \texttt {tanh} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The convention is action-bound-clean but selected success remains below the tanh base action.}}{25}{table.18}\protected@file@percent }
104
- \newlabel{tab:ctt-base-context-obs-test-tanh-rollout}{{18}{25}{Bounded \texttt {tanh} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The convention is action-bound-clean but selected success remains below the tanh base action}{table.18}{}}
105
- \@writefile{lot}{\contentsline {table}{\numberline {19}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{26}{table.19}\protected@file@percent }
106
- \newlabel{tab:ctt-val-rollout}{{19}{26}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.19}{}}
107
- \@writefile{lot}{\contentsline {table}{\numberline {20}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{27}{table.20}\protected@file@percent }
108
- \newlabel{tab:ctt-base-context-obs-val-rollout}{{20}{27}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.20}{}}
109
- \@writefile{lot}{\contentsline {table}{\numberline {21}{\ignorespaces Measured validation rollout for the deterministic RGB object-layout chart token, across three train seeds, K=8. The proxy gate passes by mean positive distance, but measured rollout does not improve over the RGB-stat validation row.}}{28}{table.21}\protected@file@percent }
110
- \newlabel{tab:ctt-base-context-obj-val-rollout}{{21}{28}{Measured validation rollout for the deterministic RGB object-layout chart token, across three train seeds, K=8. The proxy gate passes by mean positive distance, but measured rollout does not improve over the RGB-stat validation row}{table.21}{}}
111
- \@writefile{lot}{\contentsline {table}{\numberline {22}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle exceeds 50\%, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{29}{table.22}\protected@file@percent }
112
- \newlabel{tab:ctt-test-rollout}{{22}{29}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle exceeds 50\%, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.22}{}}
113
- \@writefile{lot}{\contentsline {table}{\numberline {23}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{30}{table.23}\protected@file@percent }
114
- \newlabel{tab:ctt-base-context-obs-test-rollout}{{23}{30}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.23}{}}
115
- \@writefile{lot}{\contentsline {table}{\numberline {24}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{30}{table.24}\protected@file@percent }
116
- \newlabel{tab:ctt-dominance}{{24}{30}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.24}{}}
117
- \@writefile{lot}{\contentsline {table}{\numberline {25}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains a partial selector diagnostic.}}{30}{table.25}\protected@file@percent }
118
- \newlabel{tab:ctt-learned-dominance}{{25}{30}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains a partial selector diagnostic}{table.25}{}}
119
- \@writefile{lot}{\contentsline {table}{\numberline {26}{\ignorespaces Best clipped-convention validation-calibrated dominance diagnostic: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{30}{table.26}\protected@file@percent }
120
- \newlabel{tab:ctt-base-context-obs-learned-dominance}{{26}{30}{Best clipped-convention validation-calibrated dominance diagnostic: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.26}{}}
121
- \@writefile{lot}{\contentsline {table}{\numberline {27}{\ignorespaces Bounded-tanh selector diagnostic over already measured candidates. The learned context+tangent selector is fit on validation tanh rows and evaluated once on held-out test tanh rows. It is action-bound-clean, but not a train-clean deployment selector.}}{31}{table.27}\protected@file@percent }
122
- \newlabel{tab:ctt-base-context-obs-tanh-learned-dominance}{{27}{31}{Bounded-tanh selector diagnostic over already measured candidates. The learned context+tangent selector is fit on validation tanh rows and evaluated once on held-out test tanh rows. It is action-bound-clean, but not a train-clean deployment selector}{table.27}{}}
123
- \@writefile{lot}{\contentsline {table}{\numberline {28}{\ignorespaces Train-calibrated bounded-tanh selector diagnostic. Calibration uses train-split tanh measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test tanh rows.}}{31}{table.28}\protected@file@percent }
124
- \newlabel{tab:ctt-base-context-obs-tanh-learned-train-dominance}{{28}{31}{Train-calibrated bounded-tanh selector diagnostic. Calibration uses train-split tanh measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test tanh rows}{table.28}{}}
125
- \@writefile{lot}{\contentsline {table}{\numberline {29}{\ignorespaces Train-calibrated \texttt {env\_clip} selector diagnostic. Calibration uses train-split \texttt {env\_clip} measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test \texttt {env\_clip} rows.}}{31}{table.29}\protected@file@percent }
126
- \newlabel{tab:ctt-base-context-obs-envclip-learned-train-dominance}{{29}{31}{Train-calibrated \texttt {env\_clip} selector diagnostic. Calibration uses train-split \texttt {env\_clip} measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test \texttt {env\_clip} rows}{table.29}{}}
127
- \@writefile{lot}{\contentsline {table}{\numberline {30}{\ignorespaces Train-calibrated \texttt {env\_clip} source-evidence selector diagnostic. This selector may read train-only source-chart positive/negative tangent statistics because \textsc {CTT}{} proposals are transported from measured train positive source tangents; it does not read validation/test outcomes.}}{31}{table.30}\protected@file@percent }
128
- \newlabel{tab:ctt-base-context-obs-envclip-source-evidence-dominance}{{30}{31}{Train-calibrated \texttt {env\_clip} source-evidence selector diagnostic. This selector may read train-only source-chart positive/negative tangent statistics because \ctt {} proposals are transported from measured train positive source tangents; it does not read validation/test outcomes}{table.30}{}}
129
- \@writefile{lot}{\contentsline {table}{\numberline {31}{\ignorespaces Held-out test \texttt {env\_clip} measured rollout at K=16. The table is generated by \texttt {scripts/eval\_metrics.py}; candidates are actually rolled out, so \ensuremath {\mathrm {OutcomePTR}}{}, SupportGap, and SelectorRegret are measured rather than proxy quantities.}}{32}{table.31}\protected@file@percent }
130
- \newlabel{tab:ctt-base-context-obs-envclip-k16-test-rollout}{{31}{32}{Held-out test \texttt {env\_clip} measured rollout at K=16. The table is generated by \texttt {scripts/eval\_metrics.py}; candidates are actually rolled out, so \outcomeptr {}, SupportGap, and SelectorRegret are measured rather than proxy quantities}{table.31}{}}
131
- \@writefile{lot}{\contentsline {table}{\numberline {32}{\ignorespaces Train-calibrated lower-confidence dominance fallback on the same held-out K=16 \texttt {env\_clip} measured rows. The conformal residual quantile and threshold are fit on train-calibration rows only. The artifact now reports action-bound unsafe execution and within-chart pairwise causal calibration error.}}{32}{table.32}\protected@file@percent }
132
- \newlabel{tab:ctt-base-context-obs-envclip-k16-lcb-dominance}{{32}{32}{Train-calibrated lower-confidence dominance fallback on the same held-out K=16 \texttt {env\_clip} measured rows. The conformal residual quantile and threshold are fit on train-calibration rows only. The artifact now reports action-bound unsafe execution and within-chart pairwise causal calibration error}{table.32}{}}
133
- \@writefile{lot}{\contentsline {table}{\numberline {33}{\ignorespaces Best current train-calibrated \texttt {env\_clip} K=16 selector diagnostic. Calibration uses only train-split K=16 measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test K=16 rows.}}{32}{table.33}\protected@file@percent }
134
- \newlabel{tab:ctt-base-context-obs-envclip-k16-learned-train-dominance}{{33}{32}{Best current train-calibrated \texttt {env\_clip} K=16 selector diagnostic. Calibration uses only train-split K=16 measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test K=16 rows}{table.33}{}}
135
- \@writefile{lot}{\contentsline {table}{\numberline {34}{\ignorespaces Measured outcome acceptance gate for the current K=16 \texttt {env\_clip} CTT result. The gate combines the measured rollout support artifact with the best train-clean K=16 selector and makes explicit which Part-F bars are still unmet.}}{33}{table.34}\protected@file@percent }
136
- \newlabel{tab:ctt-outcome-acceptance-gate}{{34}{33}{Measured outcome acceptance gate for the current K=16 \texttt {env\_clip} CTT result. The gate combines the measured rollout support artifact with the best train-clean K=16 selector and makes explicit which Part-F bars are still unmet}{table.34}{}}
137
- \@writefile{lot}{\contentsline {table}{\numberline {35}{\ignorespaces Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate.}}{33}{table.35}\protected@file@percent }
138
- \newlabel{tab:ctt-base-context-obs-learned-train-dominance}{{35}{33}{Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate}{table.35}{}}
139
- \@writefile{lot}{\contentsline {table}{\numberline {36}{\ignorespaces Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table\nobreakspace {}\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector.}}{33}{table.36}\protected@file@percent }
140
- \newlabel{tab:ctt-base-context-obs-nonlinear-train-dominance}{{36}{33}{Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table~\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector}{table.36}{}}
141
- \@writefile{lot}{\contentsline {table}{\numberline {37}{\ignorespaces Claim-to-artifact audit for this draft. The audit is generated by \texttt {scripts/audit\_ctt\_paper\_artifacts.py}. Warnings track the advisor's full run-contract fields such as per-run Markdown reports and logs; the current workspace policy keeps persistent prose consolidated in \texttt {README.md}.}}{33}{table.37}\protected@file@percent }
142
- \newlabel{tab:paper-ctt-artifact-audit}{{37}{33}{Claim-to-artifact audit for this draft. The audit is generated by \texttt {scripts/audit\_ctt\_paper\_artifacts.py}. Warnings track the advisor's full run-contract fields such as per-run Markdown reports and logs; the current workspace policy keeps persistent prose consolidated in \texttt {README.md}}{table.37}{}}
143
- \gdef \@abspage@last{33}
 
 
 
 
80
  \bibcite{singh2026bokbo}{7}
81
  \bibcite{tao2024maniskill3}{8}
82
  \bibcite{zhang2024vlabench}{9}
 
83
  \@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{15}{section.10}\protected@file@percent }
84
+ \bibcite{zhao2026verispace}{10}
85
+ \@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces No-clipping validation refresh for \texttt {base\_context\_obs}, K=8. This replays decoded raw actions from restored states and records action-bound validity labels before any clipping. The labels are action-space violations, not collision/contact outcomes.}}{17}{table.9}\protected@file@percent }
86
+ \newlabel{tab:ctt-base-context-obs-val-noclip-rollout}{{9}{17}{No-clipping validation refresh for \texttt {base\_context\_obs}, K=8. This replays decoded raw actions from restored states and records action-bound validity labels before any clipping. The labels are action-space violations, not collision/contact outcomes}{table.9}{}}
87
+ \@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces No-clipping held-out test refresh for \texttt {base\_context\_obs}, K=8. The proposal oracle remains nonzero, but selected success does not improve over the raw-replay base and almost all known labels indicate action-space bound violations.}}{18}{table.10}\protected@file@percent }
88
+ \newlabel{tab:ctt-base-context-obs-test-noclip-rollout}{{10}{18}{No-clipping held-out test refresh for \texttt {base\_context\_obs}, K=8. The proposal oracle remains nonzero, but selected success does not improve over the raw-replay base and almost all known labels indicate action-space bound violations}{table.10}{}}
89
+ \@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Scaled raw-action validation refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. This is a diagnostic convention suggested by the action-bound audit, not a final action representation.}}{19}{table.11}\protected@file@percent }
90
+ \newlabel{tab:ctt-base-context-obs-val-scaled-rollout}{{11}{19}{Scaled raw-action validation refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. This is a diagnostic convention suggested by the action-bound audit, not a final action representation}{table.11}{}}
91
+ \@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Scaled raw-action held-out test refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. The global scale improves base action-bound validity but does not preserve proposal support.}}{20}{table.12}\protected@file@percent }
92
+ \newlabel{tab:ctt-base-context-obs-test-scaled-rollout}{{12}{20}{Scaled raw-action held-out test refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. The global scale improves base action-bound validity but does not preserve proposal support}{table.12}{}}
93
+ \@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Per-dimension train-max scaled validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The scale vector is fit only from the train split action-bound audit.}}{21}{table.13}\protected@file@percent }
94
+ \newlabel{tab:ctt-base-context-obs-val-perdim-rollout}{{13}{21}{Per-dimension train-max scaled validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The scale vector is fit only from the train split action-bound audit}{table.13}{}}
95
+ \@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Per-dimension train-max scaled held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This diagnostic tests whether per-dimension max-fit scaling preserves more support than the global scale.}}{22}{table.14}\protected@file@percent }
96
+ \newlabel{tab:ctt-base-context-obs-test-perdim-rollout}{{14}{22}{Per-dimension train-max scaled held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This diagnostic tests whether per-dimension max-fit scaling preserves more support than the global scale}{table.14}{}}
97
+ \@writefile{lot}{\contentsline {table}{\numberline {15}{\ignorespaces Explicit \texttt {env\_clip} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The declared decoder convention clips decoded controls to action-space bounds before validity checks, so action-bound labels measure the declared convention rather than silent simulator clipping.}}{23}{table.15}\protected@file@percent }
98
+ \newlabel{tab:ctt-base-context-obs-val-envclip-rollout}{{15}{23}{Explicit \texttt {env\_clip} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The declared decoder convention clips decoded controls to action-space bounds before validity checks, so action-bound labels measure the declared convention rather than silent simulator clipping}{table.15}{}}
99
+ \@writefile{lot}{\contentsline {table}{\numberline {16}{\ignorespaces Explicit \texttt {env\_clip} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This convention is action-bound-clean and preserves more proposal support than bounded \texttt {tanh}, but score-only selection remains below base.}}{24}{table.16}\protected@file@percent }
100
+ \newlabel{tab:ctt-base-context-obs-test-envclip-rollout}{{16}{24}{Explicit \texttt {env\_clip} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This convention is action-bound-clean and preserves more proposal support than bounded \texttt {tanh}, but score-only selection remains below base}{table.16}{}}
101
+ \@writefile{lot}{\contentsline {table}{\numberline {17}{\ignorespaces Bounded \texttt {tanh} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This action convention maps decoded controls into finite action bounds before validity checks.}}{25}{table.17}\protected@file@percent }
102
+ \newlabel{tab:ctt-base-context-obs-val-tanh-rollout}{{17}{25}{Bounded \texttt {tanh} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This action convention maps decoded controls into finite action bounds before validity checks}{table.17}{}}
103
+ \@writefile{lot}{\contentsline {table}{\numberline {18}{\ignorespaces Bounded \texttt {tanh} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The convention is action-bound-clean but selected success remains below the tanh base action.}}{26}{table.18}\protected@file@percent }
104
+ \newlabel{tab:ctt-base-context-obs-test-tanh-rollout}{{18}{26}{Bounded \texttt {tanh} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The convention is action-bound-clean but selected success remains below the tanh base action}{table.18}{}}
105
+ \@writefile{lot}{\contentsline {table}{\numberline {19}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{27}{table.19}\protected@file@percent }
106
+ \newlabel{tab:ctt-val-rollout}{{19}{27}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.19}{}}
107
+ \@writefile{lot}{\contentsline {table}{\numberline {20}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{28}{table.20}\protected@file@percent }
108
+ \newlabel{tab:ctt-base-context-obs-val-rollout}{{20}{28}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.20}{}}
109
+ \@writefile{lot}{\contentsline {table}{\numberline {21}{\ignorespaces Measured validation rollout for the deterministic RGB object-layout chart token, across three train seeds, K=8. The proxy gate passes by mean positive distance, but measured rollout does not improve over the RGB-stat validation row.}}{29}{table.21}\protected@file@percent }
110
+ \newlabel{tab:ctt-base-context-obj-val-rollout}{{21}{29}{Measured validation rollout for the deterministic RGB object-layout chart token, across three train seeds, K=8. The proxy gate passes by mean positive distance, but measured rollout does not improve over the RGB-stat validation row}{table.21}{}}
111
+ \@writefile{lot}{\contentsline {table}{\numberline {22}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle exceeds 50\%, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{30}{table.22}\protected@file@percent }
112
+ \newlabel{tab:ctt-test-rollout}{{22}{30}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle exceeds 50\%, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.22}{}}
113
+ \@writefile{lot}{\contentsline {table}{\numberline {23}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{31}{table.23}\protected@file@percent }
114
+ \newlabel{tab:ctt-base-context-obs-test-rollout}{{23}{31}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.23}{}}
115
+ \@writefile{lot}{\contentsline {table}{\numberline {24}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{31}{table.24}\protected@file@percent }
116
+ \newlabel{tab:ctt-dominance}{{24}{31}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.24}{}}
117
+ \@writefile{lot}{\contentsline {table}{\numberline {25}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains a partial selector diagnostic.}}{31}{table.25}\protected@file@percent }
118
+ \newlabel{tab:ctt-learned-dominance}{{25}{31}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains a partial selector diagnostic}{table.25}{}}
119
+ \@writefile{lot}{\contentsline {table}{\numberline {26}{\ignorespaces Best clipped-convention validation-calibrated dominance diagnostic: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{31}{table.26}\protected@file@percent }
120
+ \newlabel{tab:ctt-base-context-obs-learned-dominance}{{26}{31}{Best clipped-convention validation-calibrated dominance diagnostic: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.26}{}}
121
+ \@writefile{lot}{\contentsline {table}{\numberline {27}{\ignorespaces Bounded-tanh selector diagnostic over already measured candidates. The learned context+tangent selector is fit on validation tanh rows and evaluated once on held-out test tanh rows. It is action-bound-clean, but not a train-clean deployment selector.}}{32}{table.27}\protected@file@percent }
122
+ \newlabel{tab:ctt-base-context-obs-tanh-learned-dominance}{{27}{32}{Bounded-tanh selector diagnostic over already measured candidates. The learned context+tangent selector is fit on validation tanh rows and evaluated once on held-out test tanh rows. It is action-bound-clean, but not a train-clean deployment selector}{table.27}{}}
123
+ \@writefile{lot}{\contentsline {table}{\numberline {28}{\ignorespaces Train-calibrated bounded-tanh selector diagnostic. Calibration uses train-split tanh measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test tanh rows.}}{32}{table.28}\protected@file@percent }
124
+ \newlabel{tab:ctt-base-context-obs-tanh-learned-train-dominance}{{28}{32}{Train-calibrated bounded-tanh selector diagnostic. Calibration uses train-split tanh measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test tanh rows}{table.28}{}}
125
+ \@writefile{lot}{\contentsline {table}{\numberline {29}{\ignorespaces Train-calibrated \texttt {env\_clip} selector diagnostic. Calibration uses train-split \texttt {env\_clip} measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test \texttt {env\_clip} rows.}}{32}{table.29}\protected@file@percent }
126
+ \newlabel{tab:ctt-base-context-obs-envclip-learned-train-dominance}{{29}{32}{Train-calibrated \texttt {env\_clip} selector diagnostic. Calibration uses train-split \texttt {env\_clip} measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test \texttt {env\_clip} rows}{table.29}{}}
127
+ \@writefile{lot}{\contentsline {table}{\numberline {30}{\ignorespaces Train-calibrated \texttt {env\_clip} source-evidence selector diagnostic. This selector may read train-only source-chart positive/negative tangent statistics because \textsc {CTT}{} proposals are transported from measured train positive source tangents; it does not read validation/test outcomes.}}{32}{table.30}\protected@file@percent }
128
+ \newlabel{tab:ctt-base-context-obs-envclip-source-evidence-dominance}{{30}{32}{Train-calibrated \texttt {env\_clip} source-evidence selector diagnostic. This selector may read train-only source-chart positive/negative tangent statistics because \ctt {} proposals are transported from measured train positive source tangents; it does not read validation/test outcomes}{table.30}{}}
129
+ \@writefile{lot}{\contentsline {table}{\numberline {31}{\ignorespaces Held-out test \texttt {env\_clip} measured rollout at K=16. The table is generated by \texttt {scripts/eval\_metrics.py}; candidates are actually rolled out, so \ensuremath {\mathrm {OutcomePTR}}{}, SupportGap, and SelectorRegret are measured rather than proxy quantities.}}{33}{table.31}\protected@file@percent }
130
+ \newlabel{tab:ctt-base-context-obs-envclip-k16-test-rollout}{{31}{33}{Held-out test \texttt {env\_clip} measured rollout at K=16. The table is generated by \texttt {scripts/eval\_metrics.py}; candidates are actually rolled out, so \outcomeptr {}, SupportGap, and SelectorRegret are measured rather than proxy quantities}{table.31}{}}
131
+ \@writefile{lot}{\contentsline {table}{\numberline {32}{\ignorespaces Train-calibrated lower-confidence dominance fallback on the same held-out K=16 \texttt {env\_clip} measured rows. The conformal residual quantile and threshold are fit on train-calibration rows only. The artifact now reports action-bound unsafe execution and within-chart pairwise causal calibration error.}}{33}{table.32}\protected@file@percent }
132
+ \newlabel{tab:ctt-base-context-obs-envclip-k16-lcb-dominance}{{32}{33}{Train-calibrated lower-confidence dominance fallback on the same held-out K=16 \texttt {env\_clip} measured rows. The conformal residual quantile and threshold are fit on train-calibration rows only. The artifact now reports action-bound unsafe execution and within-chart pairwise causal calibration error}{table.32}{}}
133
+ \@writefile{lot}{\contentsline {table}{\numberline {33}{\ignorespaces Best current train-calibrated \texttt {env\_clip} K=16 selector diagnostic. Calibration uses only train-split K=16 measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test K=16 rows.}}{33}{table.33}\protected@file@percent }
134
+ \newlabel{tab:ctt-base-context-obs-envclip-k16-learned-train-dominance}{{33}{33}{Best current train-calibrated \texttt {env\_clip} K=16 selector diagnostic. Calibration uses only train-split K=16 measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test K=16 rows}{table.33}{}}
135
+ \@writefile{lot}{\contentsline {table}{\numberline {34}{\ignorespaces Selector diagnostic sweep summary generated from completed selector run artifacts. For each action convention, the table reports the best row by held-out selected success while retaining all candidate rows in the \texttt {metrics.json} artifact. The comparison prevents choosing the bounded \texttt {tanh} selected-success row as the main result when its proposal support is much lower than K=16 \texttt {env\_clip}.}}{34}{table.34}\protected@file@percent }
136
+ \newlabel{tab:ctt-selector-diagnostic-sweep}{{34}{34}{Selector diagnostic sweep summary generated from completed selector run artifacts. For each action convention, the table reports the best row by held-out selected success while retaining all candidate rows in the \texttt {metrics.json} artifact. The comparison prevents choosing the bounded \texttt {tanh} selected-success row as the main result when its proposal support is much lower than K=16 \texttt {env\_clip}}{table.34}{}}
137
+ \@writefile{lot}{\contentsline {table}{\numberline {35}{\ignorespaces Measured outcome acceptance gate for the current K=16 \texttt {env\_clip} CTT result. The gate combines the measured rollout support artifact with the best train-clean K=16 selector and makes explicit which Part-F bars are still unmet.}}{34}{table.35}\protected@file@percent }
138
+ \newlabel{tab:ctt-outcome-acceptance-gate}{{35}{34}{Measured outcome acceptance gate for the current K=16 \texttt {env\_clip} CTT result. The gate combines the measured rollout support artifact with the best train-clean K=16 selector and makes explicit which Part-F bars are still unmet}{table.35}{}}
139
+ \@writefile{lot}{\contentsline {table}{\numberline {36}{\ignorespaces Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate.}}{34}{table.36}\protected@file@percent }
140
+ \newlabel{tab:ctt-base-context-obs-learned-train-dominance}{{36}{34}{Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate}{table.36}{}}
141
+ \@writefile{lot}{\contentsline {table}{\numberline {37}{\ignorespaces Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table\nobreakspace {}\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector.}}{34}{table.37}\protected@file@percent }
142
+ \newlabel{tab:ctt-base-context-obs-nonlinear-train-dominance}{{37}{34}{Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table~\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector}{table.37}{}}
143
+ \@writefile{lot}{\contentsline {table}{\numberline {38}{\ignorespaces Claim-to-artifact audit for this draft. The audit is generated by \texttt {scripts/audit\_ctt\_paper\_artifacts.py}. Warnings track the advisor's full run-contract fields such as per-run Markdown reports and logs; the current workspace policy keeps persistent prose consolidated in \texttt {README.md}.}}{34}{table.38}\protected@file@percent }
144
+ \newlabel{tab:paper-ctt-artifact-audit}{{38}{34}{Claim-to-artifact audit for this draft. The audit is generated by \texttt {scripts/audit\_ctt\_paper\_artifacts.py}. Warnings track the advisor's full run-contract fields such as per-run Markdown reports and logs; the current workspace policy keeps persistent prose consolidated in \texttt {README.md}}{table.38}{}}
145
+ \gdef \@abspage@last{34}
workspace/latex/main.fdb_latexmk CHANGED
@@ -1,20 +1,20 @@
1
  # Fdb version 4
2
- ["bibtex main"] 1783138388 "main.aux" "main.bbl" "main" 1783138389 0
3
  "./references.bib" 1783037215 3572 027cd2cf27b344368db784320d235a8c ""
4
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/bibtex/bst/base/plain.bst" 1292289607 20613 bd3fbfa9f64872b81ac57a0dd2ed855f ""
5
- "main.aux" 1783138388 29956 aa94b15e0c83b0ef43f137e2aa4a3134 "pdflatex"
6
  (generated)
7
  "main.bbl"
8
  "main.blg"
9
  (rewritten before read)
10
- ["pdflatex"] 1783138388 "main.tex" "main.pdf" "main" 1783138389 0
11
  "../paper/sections/theory.tex" 1783037322 4023 e8c0485606fdfea0020faae044bf1189 ""
12
  "../runs/action_bound_audit_rgb_refs/table.tex" 1783102548 379 5a7d42a0f3c58a06c4eeeaa280615fb3 ""
13
  "../runs/ctt_base_context_obj_val_rollout_comparison/table.tex" 1783103591 2172 16a82c2b9b79cbd8d2afdfca5503a296 ""
14
  "../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex" 1783131816 380 bd815bba39d3063468fbda949042ce2e ""
15
  "../runs/ctt_base_context_obs_learned_dominance_basic_envclip_train_to_test/table.tex" 1783121047 342 4ae8602b8d1e8ca3eb0baa82d0311b8d ""
16
  "../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex" 1783130461 363 cb90b6f3f23ea450e770d398b0ab1413 ""
17
- "../runs/ctt_base_context_obs_learned_dominance_context_success_tanh_train_to_test/table.tex" 1783107116 342 9bf813f887041aededfb26367a196a15 ""
18
  "../runs/ctt_base_context_obs_learned_dominance_context_tangent_success_tanh_val_to_test/table.tex" 1783105123 342 dd9a1e5b4a221f68ef356cf5d2587225 ""
19
  "../runs/ctt_base_context_obs_learned_dominance_context_val_to_test/table.tex" 1783094939 342 289b441acb02d293402c3de06a7e2f7a ""
20
  "../runs/ctt_base_context_obs_learned_dominance_source_envclip_train_to_test/table.tex" 1783122702 342 289ce6968fc4a468108ffcc3600b954f ""
@@ -23,13 +23,13 @@
23
  "../runs/ctt_base_context_obs_test_envclip_k16_rollout_comparison/table.tex" 1783123582 2544 3e7bc605cc803710efd73a65f6ae9c2c ""
24
  "../runs/ctt_base_context_obs_test_envclip_rollout_comparison/table.tex" 1783120980 2516 76a70f818600a600a645ca511f8631a0 ""
25
  "../runs/ctt_base_context_obs_test_noclip_rollout_comparison/table.tex" 1783103591 2514 53eb0178d1f721f8d1d39db3b88e715d ""
26
- "../runs/ctt_base_context_obs_test_perdim_trainmax_rollout_comparison/table.tex" 1783118633 2523 901a0e931d1b515e9dc956b081b93b78 ""
27
  "../runs/ctt_base_context_obs_test_rollout_comparison/table.tex" 1783103578 2170 0ea3761f3f17352fa51b5f8b4979fb70 ""
28
  "../runs/ctt_base_context_obs_test_scaled0215_rollout_comparison/table.tex" 1783103317 2523 835503f4c8c1c02ad3e6761d733eef19 ""
29
  "../runs/ctt_base_context_obs_test_tanh_rollout_comparison/table.tex" 1783104884 2518 7c061ede8cc8e34656555253887a5e0c ""
30
  "../runs/ctt_base_context_obs_val_envclip_rollout_comparison/table.tex" 1783120978 2516 59fd79135a4ada5d2783b6937fb1b494 ""
31
  "../runs/ctt_base_context_obs_val_noclip_rollout_comparison/table.tex" 1783103592 2516 efab11865e5c18448639a1adb45fc9f7 ""
32
- "../runs/ctt_base_context_obs_val_perdim_trainmax_rollout_comparison/table.tex" 1783118631 2523 ab8f8d646407958b1d62fa9cac540d97 ""
33
  "../runs/ctt_base_context_obs_val_rollout_comparison/table.tex" 1783103576 2170 ada38b0197c9b24996df6ca37ebf8ce6 ""
34
  "../runs/ctt_base_context_obs_val_scaled0215_rollout_comparison/table.tex" 1783103295 2522 dfa68ab52072ac5bcb2d63907e5bf8ca ""
35
  "../runs/ctt_base_context_obs_val_tanh_rollout_comparison/table.tex" 1783104884 2517 db488fba06c06ee1ffcf630017eeef41 ""
@@ -37,11 +37,12 @@
37
  "../runs/ctt_learned_dominance_val_to_test/table.tex" 1783051115 342 1951fdb40cb81709d46ad2391233a02b ""
38
  "../runs/ctt_outcome_acceptance_gate/table.tex" 1783138334 749 116491f873dcdfa3ed98288748e0c899 ""
39
  "../runs/ctt_residual_smoke_proxy/table.tex" 1783037068 479 a3687dfb2376b4f4dbfc05bec09c73c2 ""
 
40
  "../runs/ctt_test_rollout_comparison/table.tex" 1783103576 2171 9cc76279db5d4ffeb30bd3108263eda6 ""
41
  "../runs/ctt_val_proxy_comparison/table.tex" 1783137797 1201 46b7a5babb84523c9075eeafcf5c151c ""
42
  "../runs/ctt_val_rollout_comparison/table.tex" 1783103576 1874 6e2344e80436f3a5295e8d9c128d776d ""
43
  "../runs/data_accounting/table.tex" 1783034968 472 a35c3885ec2fd2c61f296ca898e5ee0e ""
44
- "../runs/paper_ctt_audit/table.tex" 1783138387 332 fb05065aed21217786895f919057836e ""
45
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-ts1.enc" 1136849721 2900 1537cc8184ad1792082cd229ecc269f4 ""
46
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/map/fontname/texfonts.map" 1577235249 3524 cb3e574dea2d1052e39280babc910dc8 ""
47
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/tfm/jknappen/ec/tcrm1000.tfm" 1136768653 1536 e07581a4bb3136ece9eeb4c3ffab8233 ""
@@ -161,10 +162,10 @@
161
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/web2c/texmf.cnf" 1692823885 39639 818dd8e5a0dce3a72597a7e6bad45d55 ""
162
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map" 1692821535 5031293 07973d5e761f645996ac32ebf7b8d1f3 ""
163
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/web2c/pdftex/pdflatex.fmt" 1692821530 1386016 abd9ba6eb53c47bc5de8d672b6ac87d3 ""
164
- "main.aux" 1783138388 29956 aa94b15e0c83b0ef43f137e2aa4a3134 "pdflatex"
165
- "main.bbl" 1783138388 3120 92555a174bae4ac684748bbe3fae0b8c "bibtex main"
166
- "main.out" 1783138388 2647 6c337a545f287573528c7cb52648bc95 "pdflatex"
167
- "main.tex" 1783138367 67674 50206054a25bb3e37a4ce86484261f1c ""
168
  "tables/car_decomposition.tex" 1783032219 662 bac24c0014c2ba573860ce199b776934 ""
169
  "tables/main_results.tex" 1783004655 1061 19bfdcc844614aadefc921dae15c700d ""
170
  "tables/selector_calibration.tex" 1783011198 788 1b54a8eb88519c50862f4acf718798ab ""
 
1
  # Fdb version 4
2
+ ["bibtex main"] 1783140486 "main.aux" "main.bbl" "main" 1783140515 0
3
  "./references.bib" 1783037215 3572 027cd2cf27b344368db784320d235a8c ""
4
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/bibtex/bst/base/plain.bst" 1292289607 20613 bd3fbfa9f64872b81ac57a0dd2ed855f ""
5
+ "main.aux" 1783140515 30971 dbae25d82dde9b0ed32296120ac93190 "pdflatex"
6
  (generated)
7
  "main.bbl"
8
  "main.blg"
9
  (rewritten before read)
10
+ ["pdflatex"] 1783140514 "main.tex" "main.pdf" "main" 1783140515 0
11
  "../paper/sections/theory.tex" 1783037322 4023 e8c0485606fdfea0020faae044bf1189 ""
12
  "../runs/action_bound_audit_rgb_refs/table.tex" 1783102548 379 5a7d42a0f3c58a06c4eeeaa280615fb3 ""
13
  "../runs/ctt_base_context_obj_val_rollout_comparison/table.tex" 1783103591 2172 16a82c2b9b79cbd8d2afdfca5503a296 ""
14
  "../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex" 1783131816 380 bd815bba39d3063468fbda949042ce2e ""
15
  "../runs/ctt_base_context_obs_learned_dominance_basic_envclip_train_to_test/table.tex" 1783121047 342 4ae8602b8d1e8ca3eb0baa82d0311b8d ""
16
  "../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex" 1783130461 363 cb90b6f3f23ea450e770d398b0ab1413 ""
17
+ "../runs/ctt_base_context_obs_learned_dominance_context_success_tanh_train_to_test/table.tex" 1783139605 363 dc8396ddd5f4b542770e44d200bc5ed0 ""
18
  "../runs/ctt_base_context_obs_learned_dominance_context_tangent_success_tanh_val_to_test/table.tex" 1783105123 342 dd9a1e5b4a221f68ef356cf5d2587225 ""
19
  "../runs/ctt_base_context_obs_learned_dominance_context_val_to_test/table.tex" 1783094939 342 289b441acb02d293402c3de06a7e2f7a ""
20
  "../runs/ctt_base_context_obs_learned_dominance_source_envclip_train_to_test/table.tex" 1783122702 342 289ce6968fc4a468108ffcc3600b954f ""
 
23
  "../runs/ctt_base_context_obs_test_envclip_k16_rollout_comparison/table.tex" 1783123582 2544 3e7bc605cc803710efd73a65f6ae9c2c ""
24
  "../runs/ctt_base_context_obs_test_envclip_rollout_comparison/table.tex" 1783120980 2516 76a70f818600a600a645ca511f8631a0 ""
25
  "../runs/ctt_base_context_obs_test_noclip_rollout_comparison/table.tex" 1783103591 2514 53eb0178d1f721f8d1d39db3b88e715d ""
26
+ "../runs/ctt_base_context_obs_test_perdim_trainmax_rollout_comparison/table.tex" 1783139436 2523 901a0e931d1b515e9dc956b081b93b78 ""
27
  "../runs/ctt_base_context_obs_test_rollout_comparison/table.tex" 1783103578 2170 0ea3761f3f17352fa51b5f8b4979fb70 ""
28
  "../runs/ctt_base_context_obs_test_scaled0215_rollout_comparison/table.tex" 1783103317 2523 835503f4c8c1c02ad3e6761d733eef19 ""
29
  "../runs/ctt_base_context_obs_test_tanh_rollout_comparison/table.tex" 1783104884 2518 7c061ede8cc8e34656555253887a5e0c ""
30
  "../runs/ctt_base_context_obs_val_envclip_rollout_comparison/table.tex" 1783120978 2516 59fd79135a4ada5d2783b6937fb1b494 ""
31
  "../runs/ctt_base_context_obs_val_noclip_rollout_comparison/table.tex" 1783103592 2516 efab11865e5c18448639a1adb45fc9f7 ""
32
+ "../runs/ctt_base_context_obs_val_perdim_trainmax_rollout_comparison/table.tex" 1783139434 2523 ab8f8d646407958b1d62fa9cac540d97 ""
33
  "../runs/ctt_base_context_obs_val_rollout_comparison/table.tex" 1783103576 2170 ada38b0197c9b24996df6ca37ebf8ce6 ""
34
  "../runs/ctt_base_context_obs_val_scaled0215_rollout_comparison/table.tex" 1783103295 2522 dfa68ab52072ac5bcb2d63907e5bf8ca ""
35
  "../runs/ctt_base_context_obs_val_tanh_rollout_comparison/table.tex" 1783104884 2517 db488fba06c06ee1ffcf630017eeef41 ""
 
37
  "../runs/ctt_learned_dominance_val_to_test/table.tex" 1783051115 342 1951fdb40cb81709d46ad2391233a02b ""
38
  "../runs/ctt_outcome_acceptance_gate/table.tex" 1783138334 749 116491f873dcdfa3ed98288748e0c899 ""
39
  "../runs/ctt_residual_smoke_proxy/table.tex" 1783037068 479 a3687dfb2376b4f4dbfc05bec09c73c2 ""
40
+ "../runs/ctt_selector_diagnostic_sweep/table.tex" 1783140431 553 95e3be80df669abd25cfc0552a6d7261 ""
41
  "../runs/ctt_test_rollout_comparison/table.tex" 1783103576 2171 9cc76279db5d4ffeb30bd3108263eda6 ""
42
  "../runs/ctt_val_proxy_comparison/table.tex" 1783137797 1201 46b7a5babb84523c9075eeafcf5c151c ""
43
  "../runs/ctt_val_rollout_comparison/table.tex" 1783103576 1874 6e2344e80436f3a5295e8d9c128d776d ""
44
  "../runs/data_accounting/table.tex" 1783034968 472 a35c3885ec2fd2c61f296ca898e5ee0e ""
45
+ "../runs/paper_ctt_audit/table.tex" 1783140507 332 bbc7a0fd63377a4ed4382f5d91787f3e ""
46
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-ts1.enc" 1136849721 2900 1537cc8184ad1792082cd229ecc269f4 ""
47
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/map/fontname/texfonts.map" 1577235249 3524 cb3e574dea2d1052e39280babc910dc8 ""
48
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/tfm/jknappen/ec/tcrm1000.tfm" 1136768653 1536 e07581a4bb3136ece9eeb4c3ffab8233 ""
 
162
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/web2c/texmf.cnf" 1692823885 39639 818dd8e5a0dce3a72597a7e6bad45d55 ""
163
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map" 1692821535 5031293 07973d5e761f645996ac32ebf7b8d1f3 ""
164
  "/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/web2c/pdftex/pdflatex.fmt" 1692821530 1386016 abd9ba6eb53c47bc5de8d672b6ac87d3 ""
165
+ "main.aux" 1783140515 30971 dbae25d82dde9b0ed32296120ac93190 "pdflatex"
166
+ "main.bbl" 1783140486 3120 92555a174bae4ac684748bbe3fae0b8c "bibtex main"
167
+ "main.out" 1783140515 2647 6c337a545f287573528c7cb52648bc95 "pdflatex"
168
+ "main.tex" 1783140456 68989 39c3e3573f9ebe9746abf7207609b1c2 ""
169
  "tables/car_decomposition.tex" 1783032219 662 bac24c0014c2ba573860ce199b776934 ""
170
  "tables/main_results.tex" 1783004655 1061 19bfdcc844614aadefc921dae15c700d ""
171
  "tables/selector_calibration.tex" 1783011198 788 1b54a8eb88519c50862f4acf718798ab ""
workspace/latex/main.fls CHANGED
@@ -999,6 +999,17 @@ INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_tas
999
  INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
1000
  INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
1001
  INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
 
 
 
 
 
 
 
 
 
 
 
1002
  INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
1003
  INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
1004
  INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
 
999
  INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
1000
  INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
1001
  INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
1002
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1003
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1004
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1005
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1006
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1007
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1008
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1009
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1010
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1011
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1012
+ INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
1013
  INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
1014
  INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
1015
  INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
workspace/latex/main.log CHANGED
@@ -1,4 +1,4 @@
1
- This is pdfTeX, Version 3.141592653-2.6-1.40.22 (TeX Live 2021 Gentoo Linux) (preloaded format=pdflatex 2023.8.23) 4 JUL 2026 00:13
2
  entering extended mode
3
  restricted \write18 enabled.
4
  %&-line parsing enabled.
@@ -602,119 +602,119 @@ Underfull \hbox (badness 3439) in paragraph at lines 972--972
602
  (../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex)
603
  (../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_en
604
  vclip_k16_train_to_test/table.tex)
605
- (../runs/ctt_outcome_acceptance_gate/table.tex)
 
606
  (../runs/ctt_base_context_obs_learned_dominance_train_to_test/table.tex)
607
- [12]
608
  (../runs/ctt_base_context_obs_nonlinear_dominance_basic_positive_train_to_test/
609
  table.tex)
610
- Overfull \hbox (31.94885pt too wide) in paragraph at lines 1138--1142
611
  []\OT1/cmtt/m/n/10 scripts/export[]cil[]charts.py\OT1/cmr/m/n/10 , \OT1/cmtt/m/
612
  n/10 scripts/build[]data[]accounting.py\OT1/cmr/m/n/10 , and \OT1/cmtt/m/n/10 s
613
  cripts/audit[]cil[]charts.py
614
  []
615
 
616
 
617
- Overfull \hbox (127.11275pt too wide) in paragraph at lines 1144--1148
618
  []\OT1/cmtt/m/n/10 cil/chart[]features.py \OT1/cmr/m/n/10 cen-tral-izes deploym
619
  ent-visible chart fea-ture con-struc-tion, and \OT1/cmtt/m/n/10 scripts/audit[]
620
  chart[]feature[]sources.py
621
  []
622
 
623
 
624
- Overfull \hbox (104.78587pt too wide) in paragraph at lines 1148--1151
625
  []\OT1/cmtt/m/n/10 scripts/slurm/render[]six[]task[]chart[]observations.sbatch
626
  \OT1/cmr/m/n/10 and \OT1/cmtt/m/n/10 scripts/slurm/reexport[]rgb[]ref[]cil[]cha
627
  rts.sbatch
628
  []
629
 
630
 
631
- Overfull \hbox (23.72661pt too wide) in paragraph at lines 1151--1156
632
  []\OT1/cmtt/m/n/10 scripts/export[]chart[]observation[]embeddings.py \OT1/cmr/m
633
  /n/10 and \OT1/cmtt/m/n/10 scripts/export[]chart[]object[]embeddings.py
634
  []
635
 
636
 
637
- Overfull \hbox (129.91359pt too wide) in paragraph at lines 1151--1156
638
  \OT1/cmr/m/n/10 cre-ate de-ter-min-is-tic 32D RGB-stat and 64D RGB object-layou
639
  t em-bed-dings, and \OT1/cmtt/m/n/10 scripts/slurm/train[]ctt[]feature[]proxy.s
640
  batch
641
  []
642
 
643
 
644
- Overfull \hbox (7.97675pt too wide) in paragraph at lines 1156--1165
645
  []\OT1/cmtt/m/n/10 scripts/eval[]ctt[]generated[]rollout.py \OT1/cmr/m/n/10 and
646
  \OT1/cmtt/m/n/10 scripts/slurm/eval[]ctt[]generated[]rollout.sbatch
647
  []
648
 
649
 
650
- Overfull \hbox (20.61057pt too wide) in paragraph at lines 1156--1165
651
  \OT1/cmr/m/n/10 train-split cal-i-bra-tion and meta-data load-ing for deploymen
652
  t-visible chart fea-tures such as \OT1/cmtt/m/n/10 base[]context[]obs\OT1/cmr/m
653
  /n/10 .
654
  []
655
 
656
 
657
- Overfull \hbox (49.05014pt too wide) in paragraph at lines 1156--1165
658
  \OT1/cmr/m/n/10 They also record action-bound va-lid-ity la-bels and sup-port r
659
  aw-action re-play through \OT1/cmtt/m/n/10 --disable-action-clipping
660
  []
661
 
662
 
663
- Overfull \hbox (2.58342pt too wide) in paragraph at lines 1165--1169
664
  []\OT1/cmtt/m/n/10 scripts/eval[]nonlinear[]dominance[]selector.py \OT1/cmr/m/n
665
  /10 shares the chart-compatibility feature-loading path
666
  []
667
 
668
 
669
- Overfull \hbox (11.7557pt too wide) in paragraph at lines 1169--1172
670
  []\OT1/cmtt/m/n/10 scripts/audit[]action[]bounds.py \OT1/cmr/m/n/10 pro-duces \
671
  OT1/cmtt/m/n/10 runs/action[]bound[]audit[]rgb[]refs\OT1/cmr/m/n/10 , which au-
672
  dits whether
673
  []
674
 
675
  [13]
676
- Overfull \hbox (6.0625pt too wide) in paragraph at lines 1191--1194
677
  []\OT1/cmtt/m/n/10 scripts/eval[]chart[]positive[]memory[]proxy.py \OT1/cmr/m/n
678
  /10 and \OT1/cmtt/m/n/10 scripts/build[]ctt[]proxy[]comparison.py \OT1/cmr/m/n/
679
  10 gen-
680
  []
681
 
682
 
683
- Overfull \hbox (63.81772pt too wide) in paragraph at lines 1194--1199
684
  \OT1/cmr/m/n/10 min-is-tic sum-maries of \OT1/cmtt/m/n/10 delta[]action\OT1/cmr
685
  /m/n/10 ; \OT1/cmtt/m/n/10 runs/tangent[]reconstruction \OT1/cmr/m/n/10 and \OT
686
  1/cmtt/m/n/10 runs/tangent[]reconstruction[]rgb[]refs
687
  []
688
 
689
 
690
- Overfull \hbox (61.283pt too wide) in paragraph at lines 1199--1203
691
  []\OT1/cmtt/m/n/10 scripts/audit[]cil[]charts.py \OT1/cmr/m/n/10 writes the lea
692
  k-age re-ports \OT1/cmtt/m/n/10 runs/leakage[]audit \OT1/cmr/m/n/10 and \OT1/cm
693
  tt/m/n/10 runs/leakage[]audit[]rgb[]refs\OT1/cmr/m/n/10 ,
694
  []
695
 
696
 
697
- Overfull \hbox (20.3325pt too wide) in paragraph at lines 1203--1206
698
  []\OT1/cmtt/m/n/10 scripts/train[]utility[]energy.py \OT1/cmr/m/n/10 and \OT1/c
699
  mtt/m/n/10 scripts/calibrate[]dominance.py \OT1/cmr/m/n/10 im-ple-ment the util
700
  -ity/scoring
701
  []
702
 
703
 
704
- Overfull \hbox (3.63927pt too wide) in paragraph at lines 1212--1216
705
  []\OT1/cmtt/m/n/10 scripts/backfill[]paper[]run[]artifacts.py \OT1/cmr/m/n/10 t
706
  rans-par-ently back-fills non-Markdown run meta-data such
707
  []
708
 
709
 
710
- Underfull \hbox (badness 10000) in paragraph at lines 1224--1224
711
- []\OT1/cmr/m/n/10 Table 37: []Claim-to-artifact au-dit for this draft. The au-d
712
  it is gen-er-ated by
713
  []
714
 
715
- (../runs/paper_ctt_audit/table.tex) [14] (./main.bbl) [15] [16] [17] [18]
716
  [19] [20] [21] [22] [23] [24] [25] [26] [27] [28] [29] [30] [31] [32] [33]
717
- (./main.aux)
718
 
719
  LaTeX Font Warning: Some font shapes were not available, defaults substituted.
720
 
@@ -722,13 +722,13 @@ Package rerunfilecheck Info: File `main.out' has not changed.
722
  (rerunfilecheck) Checksum: 6C337A545F287573528C7CB52648BC95;2647.
723
  )
724
  Here is how much of TeX's memory you used:
725
- 10288 strings out of 480884
726
- 159129 string characters out of 5900692
727
- 721665 words of memory out of 5000000
728
- 27159 multiletter control sequences out of 15000+600000
729
  414764 words of font info for 71 fonts, out of 8000000 for 9000
730
  36 hyphenation exceptions out of 8191
731
- 71i,10n,74p,608b,392s stack positions out of 5000i,500n,10000p,200000b,80000s
732
  {/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/font
733
  s/enc/dvips/cm-super/cm-super-ts1.enc}</cvmfs/soft.computecanada.ca/gentoo/2023
734
  /x86-64-v3/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmbx10.pfb></cvm
@@ -769,10 +769,10 @@ cm/cmtt9.pfb></cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texm
769
  f-dist/fonts/type1/public/amsfonts/symbols/msbm10.pfb></cvmfs/soft.computecanad
770
  a.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/type1/public/cm-super/sfr
771
  m1000.pfb>
772
- Output written on main.pdf (33 pages, 435784 bytes).
773
  PDF statistics:
774
- 454 PDF objects out of 1000 (max. 8388607)
775
- 388 compressed objects within 4 object streams
776
- 116 named destinations out of 1000 (max. 500000)
777
  137 words of extra memory for PDF output out of 10000 (max. 10000000)
778
 
 
1
+ This is pdfTeX, Version 3.141592653-2.6-1.40.22 (TeX Live 2021 Gentoo Linux) (preloaded format=pdflatex 2023.8.23) 4 JUL 2026 00:48
2
  entering extended mode
3
  restricted \write18 enabled.
4
  %&-line parsing enabled.
 
602
  (../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex)
603
  (../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_en
604
  vclip_k16_train_to_test/table.tex)
605
+ (../runs/ctt_selector_diagnostic_sweep/table.tex)
606
+ (../runs/ctt_outcome_acceptance_gate/table.tex) [12]
607
  (../runs/ctt_base_context_obs_learned_dominance_train_to_test/table.tex)
 
608
  (../runs/ctt_base_context_obs_nonlinear_dominance_basic_positive_train_to_test/
609
  table.tex)
610
+ Overfull \hbox (31.94885pt too wide) in paragraph at lines 1162--1166
611
  []\OT1/cmtt/m/n/10 scripts/export[]cil[]charts.py\OT1/cmr/m/n/10 , \OT1/cmtt/m/
612
  n/10 scripts/build[]data[]accounting.py\OT1/cmr/m/n/10 , and \OT1/cmtt/m/n/10 s
613
  cripts/audit[]cil[]charts.py
614
  []
615
 
616
 
617
+ Overfull \hbox (127.11275pt too wide) in paragraph at lines 1168--1172
618
  []\OT1/cmtt/m/n/10 cil/chart[]features.py \OT1/cmr/m/n/10 cen-tral-izes deploym
619
  ent-visible chart fea-ture con-struc-tion, and \OT1/cmtt/m/n/10 scripts/audit[]
620
  chart[]feature[]sources.py
621
  []
622
 
623
 
624
+ Overfull \hbox (104.78587pt too wide) in paragraph at lines 1172--1175
625
  []\OT1/cmtt/m/n/10 scripts/slurm/render[]six[]task[]chart[]observations.sbatch
626
  \OT1/cmr/m/n/10 and \OT1/cmtt/m/n/10 scripts/slurm/reexport[]rgb[]ref[]cil[]cha
627
  rts.sbatch
628
  []
629
 
630
 
631
+ Overfull \hbox (23.72661pt too wide) in paragraph at lines 1175--1180
632
  []\OT1/cmtt/m/n/10 scripts/export[]chart[]observation[]embeddings.py \OT1/cmr/m
633
  /n/10 and \OT1/cmtt/m/n/10 scripts/export[]chart[]object[]embeddings.py
634
  []
635
 
636
 
637
+ Overfull \hbox (129.91359pt too wide) in paragraph at lines 1175--1180
638
  \OT1/cmr/m/n/10 cre-ate de-ter-min-is-tic 32D RGB-stat and 64D RGB object-layou
639
  t em-bed-dings, and \OT1/cmtt/m/n/10 scripts/slurm/train[]ctt[]feature[]proxy.s
640
  batch
641
  []
642
 
643
 
644
+ Overfull \hbox (7.97675pt too wide) in paragraph at lines 1180--1189
645
  []\OT1/cmtt/m/n/10 scripts/eval[]ctt[]generated[]rollout.py \OT1/cmr/m/n/10 and
646
  \OT1/cmtt/m/n/10 scripts/slurm/eval[]ctt[]generated[]rollout.sbatch
647
  []
648
 
649
 
650
+ Overfull \hbox (20.61057pt too wide) in paragraph at lines 1180--1189
651
  \OT1/cmr/m/n/10 train-split cal-i-bra-tion and meta-data load-ing for deploymen
652
  t-visible chart fea-tures such as \OT1/cmtt/m/n/10 base[]context[]obs\OT1/cmr/m
653
  /n/10 .
654
  []
655
 
656
 
657
+ Overfull \hbox (49.05014pt too wide) in paragraph at lines 1180--1189
658
  \OT1/cmr/m/n/10 They also record action-bound va-lid-ity la-bels and sup-port r
659
  aw-action re-play through \OT1/cmtt/m/n/10 --disable-action-clipping
660
  []
661
 
662
 
663
+ Overfull \hbox (2.58342pt too wide) in paragraph at lines 1189--1193
664
  []\OT1/cmtt/m/n/10 scripts/eval[]nonlinear[]dominance[]selector.py \OT1/cmr/m/n
665
  /10 shares the chart-compatibility feature-loading path
666
  []
667
 
668
 
669
+ Overfull \hbox (11.7557pt too wide) in paragraph at lines 1193--1196
670
  []\OT1/cmtt/m/n/10 scripts/audit[]action[]bounds.py \OT1/cmr/m/n/10 pro-duces \
671
  OT1/cmtt/m/n/10 runs/action[]bound[]audit[]rgb[]refs\OT1/cmr/m/n/10 , which au-
672
  dits whether
673
  []
674
 
675
  [13]
676
+ Overfull \hbox (6.0625pt too wide) in paragraph at lines 1215--1218
677
  []\OT1/cmtt/m/n/10 scripts/eval[]chart[]positive[]memory[]proxy.py \OT1/cmr/m/n
678
  /10 and \OT1/cmtt/m/n/10 scripts/build[]ctt[]proxy[]comparison.py \OT1/cmr/m/n/
679
  10 gen-
680
  []
681
 
682
 
683
+ Overfull \hbox (63.81772pt too wide) in paragraph at lines 1218--1223
684
  \OT1/cmr/m/n/10 min-is-tic sum-maries of \OT1/cmtt/m/n/10 delta[]action\OT1/cmr
685
  /m/n/10 ; \OT1/cmtt/m/n/10 runs/tangent[]reconstruction \OT1/cmr/m/n/10 and \OT
686
  1/cmtt/m/n/10 runs/tangent[]reconstruction[]rgb[]refs
687
  []
688
 
689
 
690
+ Overfull \hbox (61.283pt too wide) in paragraph at lines 1223--1227
691
  []\OT1/cmtt/m/n/10 scripts/audit[]cil[]charts.py \OT1/cmr/m/n/10 writes the lea
692
  k-age re-ports \OT1/cmtt/m/n/10 runs/leakage[]audit \OT1/cmr/m/n/10 and \OT1/cm
693
  tt/m/n/10 runs/leakage[]audit[]rgb[]refs\OT1/cmr/m/n/10 ,
694
  []
695
 
696
 
697
+ Overfull \hbox (20.3325pt too wide) in paragraph at lines 1227--1230
698
  []\OT1/cmtt/m/n/10 scripts/train[]utility[]energy.py \OT1/cmr/m/n/10 and \OT1/c
699
  mtt/m/n/10 scripts/calibrate[]dominance.py \OT1/cmr/m/n/10 im-ple-ment the util
700
  -ity/scoring
701
  []
702
 
703
 
704
+ Overfull \hbox (3.63927pt too wide) in paragraph at lines 1236--1240
705
  []\OT1/cmtt/m/n/10 scripts/backfill[]paper[]run[]artifacts.py \OT1/cmr/m/n/10 t
706
  rans-par-ently back-fills non-Markdown run meta-data such
707
  []
708
 
709
 
710
+ Underfull \hbox (badness 10000) in paragraph at lines 1248--1248
711
+ []\OT1/cmr/m/n/10 Table 38: []Claim-to-artifact au-dit for this draft. The au-d
712
  it is gen-er-ated by
713
  []
714
 
715
+ (../runs/paper_ctt_audit/table.tex) [14] (./main.bbl [15]) [16] [17] [18]
716
  [19] [20] [21] [22] [23] [24] [25] [26] [27] [28] [29] [30] [31] [32] [33]
717
+ [34] (./main.aux)
718
 
719
  LaTeX Font Warning: Some font shapes were not available, defaults substituted.
720
 
 
722
  (rerunfilecheck) Checksum: 6C337A545F287573528C7CB52648BC95;2647.
723
  )
724
  Here is how much of TeX's memory you used:
725
+ 10295 strings out of 480884
726
+ 159358 string characters out of 5900692
727
+ 727858 words of memory out of 5000000
728
+ 27161 multiletter control sequences out of 15000+600000
729
  414764 words of font info for 71 fonts, out of 8000000 for 9000
730
  36 hyphenation exceptions out of 8191
731
+ 71i,10n,74p,608b,446s stack positions out of 5000i,500n,10000p,200000b,80000s
732
  {/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/font
733
  s/enc/dvips/cm-super/cm-super-ts1.enc}</cvmfs/soft.computecanada.ca/gentoo/2023
734
  /x86-64-v3/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmbx10.pfb></cvm
 
769
  f-dist/fonts/type1/public/amsfonts/symbols/msbm10.pfb></cvmfs/soft.computecanad
770
  a.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/type1/public/cm-super/sfr
771
  m1000.pfb>
772
+ Output written on main.pdf (34 pages, 438012 bytes).
773
  PDF statistics:
774
+ 460 PDF objects out of 1000 (max. 8388607)
775
+ 393 compressed objects within 4 object streams
776
+ 118 named destinations out of 1000 (max. 500000)
777
  137 words of extra memory for PDF output out of 10000 (max. 10000000)
778
 
workspace/latex/main.pdf CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2db143e6a5e6fdbe0e42bc080e29f28a53bc3ba6cc787a6102cfd73394c9ebb5
3
- size 435784
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd3e913de31fb2f4a8c4a4c9393216a45036075bb97cc953339a6c61400872af
3
+ size 438012
workspace/latex/main.tex CHANGED
@@ -1054,6 +1054,30 @@ conclusion is sharper than the K=8 result: support is now strong enough to make
1054
  the paper story credible, train-only visual chart compatibility is useful, but
1055
  selector/utility energy is still the dominant remaining bottleneck.
1056
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1057
  \begin{table}[t]
1058
  \centering
1059
  \caption{Measured outcome acceptance gate for the current K=16
 
1054
  the paper story credible, train-only visual chart compatibility is useful, but
1055
  selector/utility energy is still the dominant remaining bottleneck.
1056
 
1057
+ \begin{table}[t]
1058
+ \centering
1059
+ \caption{Selector diagnostic sweep summary generated from completed selector
1060
+ run artifacts. For each action convention, the table reports the best row by
1061
+ held-out selected success while retaining all candidate rows in the
1062
+ \texttt{metrics.json} artifact. The comparison prevents choosing the bounded
1063
+ \texttt{tanh} selected-success row as the main result when its proposal support
1064
+ is much lower than K=16 \texttt{env\_clip}.}
1065
+ \label{tab:ctt-selector-diagnostic-sweep}
1066
+ \scriptsize
1067
+ \resizebox{\linewidth}{!}{\input{../runs/ctt_selector_diagnostic_sweep/table}}
1068
+ \end{table}
1069
+
1070
+ Table~\ref{tab:ctt-selector-diagnostic-sweep} summarizes the selector
1071
+ diagnostics without hiding the tradeoff. The K=8 bounded-\texttt{tanh} selector
1072
+ reaches the highest selected success, 38.19\%, but its proposal oracle is also
1073
+ only 38.19\%, so it is not a stronger support result. The K=8 per-dimension
1074
+ train-max convention mostly falls back to the base action and has still lower
1075
+ generated support. K=16 \texttt{env\_clip} remains the strongest support
1076
+ setting, with a 56.94\% proposal oracle and 54.86\% \outcomeptr{}@16, but it
1077
+ leaves the largest selector gap. Thus the honest conclusion is not that one
1078
+ post-processing convention solves deployment; it is that support and selection
1079
+ must both pass their gates.
1080
+
1081
  \begin{table}[t]
1082
  \centering
1083
  \caption{Measured outcome acceptance gate for the current K=16