manual-selector-sweep-paper-sync 2026-07-04T04:48:53Z workspace
Browse files- workspace/README.md +7 -0
- workspace/latex/main.aux +62 -60
- workspace/latex/main.fdb_latexmk +12 -11
- workspace/latex/main.fls +11 -0
- workspace/latex/main.log +31 -31
- workspace/latex/main.pdf +2 -2
- workspace/latex/main.tex +24 -0
workspace/README.md
CHANGED
|
@@ -292,6 +292,10 @@ Dominance and utility:
|
|
| 292 |
features. Use `--no-markdown-report` for README-only runs.
|
| 293 |
- `scripts/eval_nonlinear_dominance_selector.py`: nonlinear selector sweep.
|
| 294 |
Use `--no-markdown-report` for README-only runs.
|
|
|
|
|
|
|
|
|
|
|
|
|
| 295 |
|
| 296 |
Metric and paper artifacts:
|
| 297 |
|
|
@@ -443,6 +447,9 @@ High-value run directories:
|
|
| 443 |
- `runs/ctt_base_context_obs_learned_dominance_*_perdim_trainmax_train_to_test`:
|
| 444 |
per-dimension trainmax selector diagnostics from Slurm job `15149815`; these
|
| 445 |
are below the current K=16 `env_clip` support setting.
|
|
|
|
|
|
|
|
|
|
| 446 |
- `runs/ctt_base_context_obs_nonlinear_dominance_chartcompat_obs_*`: fixed
|
| 447 |
nonlinear selector diagnostics.
|
| 448 |
- `runs/summary_ctt.csv`: global run summary table.
|
|
|
|
| 292 |
features. Use `--no-markdown-report` for README-only runs.
|
| 293 |
- `scripts/eval_nonlinear_dominance_selector.py`: nonlinear selector sweep.
|
| 294 |
Use `--no-markdown-report` for README-only runs.
|
| 295 |
+
- `scripts/build_selector_diagnostic_sweep.py`: non-cherry-picked selector
|
| 296 |
+
diagnostic summary builder. It reads completed selector `metrics.json`
|
| 297 |
+
artifacts, keeps all candidate rows, selects the best row per action
|
| 298 |
+
convention by held-out selected success, and writes JSON/TeX/log outputs.
|
| 299 |
|
| 300 |
Metric and paper artifacts:
|
| 301 |
|
|
|
|
| 447 |
- `runs/ctt_base_context_obs_learned_dominance_*_perdim_trainmax_train_to_test`:
|
| 448 |
per-dimension trainmax selector diagnostics from Slurm job `15149815`; these
|
| 449 |
are below the current K=16 `env_clip` support setting.
|
| 450 |
+
- `runs/ctt_selector_diagnostic_sweep`: generated selector summary table used
|
| 451 |
+
by the paper to compare K=8 tanh, K=8 per-dim trainmax, and K=16 env-clip
|
| 452 |
+
selector diagnostics without cherry-picking a single row.
|
| 453 |
- `runs/ctt_base_context_obs_nonlinear_dominance_chartcompat_obs_*`: fixed
|
| 454 |
nonlinear selector diagnostics.
|
| 455 |
- `runs/summary_ctt.csv`: global run summary table.
|
workspace/latex/main.aux
CHANGED
|
@@ -80,64 +80,66 @@
|
|
| 80 |
\bibcite{singh2026bokbo}{7}
|
| 81 |
\bibcite{tao2024maniskill3}{8}
|
| 82 |
\bibcite{zhang2024vlabench}{9}
|
| 83 |
-
\bibcite{zhao2026verispace}{10}
|
| 84 |
\@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{15}{section.10}\protected@file@percent }
|
| 85 |
-
\
|
| 86 |
-
\
|
| 87 |
-
\
|
| 88 |
-
\
|
| 89 |
-
\
|
| 90 |
-
\
|
| 91 |
-
\
|
| 92 |
-
\
|
| 93 |
-
\
|
| 94 |
-
\
|
| 95 |
-
\
|
| 96 |
-
\
|
| 97 |
-
\
|
| 98 |
-
\
|
| 99 |
-
\
|
| 100 |
-
\
|
| 101 |
-
\
|
| 102 |
-
\
|
| 103 |
-
\
|
| 104 |
-
\
|
| 105 |
-
\
|
| 106 |
-
\
|
| 107 |
-
\
|
| 108 |
-
\
|
| 109 |
-
\
|
| 110 |
-
\
|
| 111 |
-
\
|
| 112 |
-
\
|
| 113 |
-
\
|
| 114 |
-
\
|
| 115 |
-
\
|
| 116 |
-
\
|
| 117 |
-
\
|
| 118 |
-
\
|
| 119 |
-
\
|
| 120 |
-
\
|
| 121 |
-
\
|
| 122 |
-
\
|
| 123 |
-
\
|
| 124 |
-
\
|
| 125 |
-
\
|
| 126 |
-
\
|
| 127 |
-
\
|
| 128 |
-
\
|
| 129 |
-
\
|
| 130 |
-
\
|
| 131 |
-
\
|
| 132 |
-
\
|
| 133 |
-
\
|
| 134 |
-
\
|
| 135 |
-
\
|
| 136 |
-
\
|
| 137 |
-
\
|
| 138 |
-
\
|
| 139 |
-
\
|
| 140 |
-
\
|
| 141 |
-
\
|
| 142 |
-
\
|
| 143 |
-
\
|
|
|
|
|
|
|
|
|
|
|
|
| 80 |
\bibcite{singh2026bokbo}{7}
|
| 81 |
\bibcite{tao2024maniskill3}{8}
|
| 82 |
\bibcite{zhang2024vlabench}{9}
|
|
|
|
| 83 |
\@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{15}{section.10}\protected@file@percent }
|
| 84 |
+
\bibcite{zhao2026verispace}{10}
|
| 85 |
+
\@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces No-clipping validation refresh for \texttt {base\_context\_obs}, K=8. This replays decoded raw actions from restored states and records action-bound validity labels before any clipping. The labels are action-space violations, not collision/contact outcomes.}}{17}{table.9}\protected@file@percent }
|
| 86 |
+
\newlabel{tab:ctt-base-context-obs-val-noclip-rollout}{{9}{17}{No-clipping validation refresh for \texttt {base\_context\_obs}, K=8. This replays decoded raw actions from restored states and records action-bound validity labels before any clipping. The labels are action-space violations, not collision/contact outcomes}{table.9}{}}
|
| 87 |
+
\@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces No-clipping held-out test refresh for \texttt {base\_context\_obs}, K=8. The proposal oracle remains nonzero, but selected success does not improve over the raw-replay base and almost all known labels indicate action-space bound violations.}}{18}{table.10}\protected@file@percent }
|
| 88 |
+
\newlabel{tab:ctt-base-context-obs-test-noclip-rollout}{{10}{18}{No-clipping held-out test refresh for \texttt {base\_context\_obs}, K=8. The proposal oracle remains nonzero, but selected success does not improve over the raw-replay base and almost all known labels indicate action-space bound violations}{table.10}{}}
|
| 89 |
+
\@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Scaled raw-action validation refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. This is a diagnostic convention suggested by the action-bound audit, not a final action representation.}}{19}{table.11}\protected@file@percent }
|
| 90 |
+
\newlabel{tab:ctt-base-context-obs-val-scaled-rollout}{{11}{19}{Scaled raw-action validation refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. This is a diagnostic convention suggested by the action-bound audit, not a final action representation}{table.11}{}}
|
| 91 |
+
\@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Scaled raw-action held-out test refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. The global scale improves base action-bound validity but does not preserve proposal support.}}{20}{table.12}\protected@file@percent }
|
| 92 |
+
\newlabel{tab:ctt-base-context-obs-test-scaled-rollout}{{12}{20}{Scaled raw-action held-out test refresh for \texttt {base\_context\_obs}, K=8, with \texttt {--execution-action-scale 0.215} and clipping disabled. The global scale improves base action-bound validity but does not preserve proposal support}{table.12}{}}
|
| 93 |
+
\@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Per-dimension train-max scaled validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The scale vector is fit only from the train split action-bound audit.}}{21}{table.13}\protected@file@percent }
|
| 94 |
+
\newlabel{tab:ctt-base-context-obs-val-perdim-rollout}{{13}{21}{Per-dimension train-max scaled validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The scale vector is fit only from the train split action-bound audit}{table.13}{}}
|
| 95 |
+
\@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Per-dimension train-max scaled held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This diagnostic tests whether per-dimension max-fit scaling preserves more support than the global scale.}}{22}{table.14}\protected@file@percent }
|
| 96 |
+
\newlabel{tab:ctt-base-context-obs-test-perdim-rollout}{{14}{22}{Per-dimension train-max scaled held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This diagnostic tests whether per-dimension max-fit scaling preserves more support than the global scale}{table.14}{}}
|
| 97 |
+
\@writefile{lot}{\contentsline {table}{\numberline {15}{\ignorespaces Explicit \texttt {env\_clip} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The declared decoder convention clips decoded controls to action-space bounds before validity checks, so action-bound labels measure the declared convention rather than silent simulator clipping.}}{23}{table.15}\protected@file@percent }
|
| 98 |
+
\newlabel{tab:ctt-base-context-obs-val-envclip-rollout}{{15}{23}{Explicit \texttt {env\_clip} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The declared decoder convention clips decoded controls to action-space bounds before validity checks, so action-bound labels measure the declared convention rather than silent simulator clipping}{table.15}{}}
|
| 99 |
+
\@writefile{lot}{\contentsline {table}{\numberline {16}{\ignorespaces Explicit \texttt {env\_clip} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This convention is action-bound-clean and preserves more proposal support than bounded \texttt {tanh}, but score-only selection remains below base.}}{24}{table.16}\protected@file@percent }
|
| 100 |
+
\newlabel{tab:ctt-base-context-obs-test-envclip-rollout}{{16}{24}{Explicit \texttt {env\_clip} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This convention is action-bound-clean and preserves more proposal support than bounded \texttt {tanh}, but score-only selection remains below base}{table.16}{}}
|
| 101 |
+
\@writefile{lot}{\contentsline {table}{\numberline {17}{\ignorespaces Bounded \texttt {tanh} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This action convention maps decoded controls into finite action bounds before validity checks.}}{25}{table.17}\protected@file@percent }
|
| 102 |
+
\newlabel{tab:ctt-base-context-obs-val-tanh-rollout}{{17}{25}{Bounded \texttt {tanh} validation refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. This action convention maps decoded controls into finite action bounds before validity checks}{table.17}{}}
|
| 103 |
+
\@writefile{lot}{\contentsline {table}{\numberline {18}{\ignorespaces Bounded \texttt {tanh} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The convention is action-bound-clean but selected success remains below the tanh base action.}}{26}{table.18}\protected@file@percent }
|
| 104 |
+
\newlabel{tab:ctt-base-context-obs-test-tanh-rollout}{{18}{26}{Bounded \texttt {tanh} held-out test refresh for \texttt {base\_context\_obs}, K=8, with clipping disabled. The convention is action-bound-clean but selected success remains below the tanh base action}{table.18}{}}
|
| 105 |
+
\@writefile{lot}{\contentsline {table}{\numberline {19}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{27}{table.19}\protected@file@percent }
|
| 106 |
+
\newlabel{tab:ctt-val-rollout}{{19}{27}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.19}{}}
|
| 107 |
+
\@writefile{lot}{\contentsline {table}{\numberline {20}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{28}{table.20}\protected@file@percent }
|
| 108 |
+
\newlabel{tab:ctt-base-context-obs-val-rollout}{{20}{28}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.20}{}}
|
| 109 |
+
\@writefile{lot}{\contentsline {table}{\numberline {21}{\ignorespaces Measured validation rollout for the deterministic RGB object-layout chart token, across three train seeds, K=8. The proxy gate passes by mean positive distance, but measured rollout does not improve over the RGB-stat validation row.}}{29}{table.21}\protected@file@percent }
|
| 110 |
+
\newlabel{tab:ctt-base-context-obj-val-rollout}{{21}{29}{Measured validation rollout for the deterministic RGB object-layout chart token, across three train seeds, K=8. The proxy gate passes by mean positive distance, but measured rollout does not improve over the RGB-stat validation row}{table.21}{}}
|
| 111 |
+
\@writefile{lot}{\contentsline {table}{\numberline {22}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle exceeds 50\%, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{30}{table.22}\protected@file@percent }
|
| 112 |
+
\newlabel{tab:ctt-test-rollout}{{22}{30}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle exceeds 50\%, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.22}{}}
|
| 113 |
+
\@writefile{lot}{\contentsline {table}{\numberline {23}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{31}{table.23}\protected@file@percent }
|
| 114 |
+
\newlabel{tab:ctt-base-context-obs-test-rollout}{{23}{31}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.23}{}}
|
| 115 |
+
\@writefile{lot}{\contentsline {table}{\numberline {24}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{31}{table.24}\protected@file@percent }
|
| 116 |
+
\newlabel{tab:ctt-dominance}{{24}{31}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.24}{}}
|
| 117 |
+
\@writefile{lot}{\contentsline {table}{\numberline {25}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains a partial selector diagnostic.}}{31}{table.25}\protected@file@percent }
|
| 118 |
+
\newlabel{tab:ctt-learned-dominance}{{25}{31}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains a partial selector diagnostic}{table.25}{}}
|
| 119 |
+
\@writefile{lot}{\contentsline {table}{\numberline {26}{\ignorespaces Best clipped-convention validation-calibrated dominance diagnostic: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{31}{table.26}\protected@file@percent }
|
| 120 |
+
\newlabel{tab:ctt-base-context-obs-learned-dominance}{{26}{31}{Best clipped-convention validation-calibrated dominance diagnostic: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.26}{}}
|
| 121 |
+
\@writefile{lot}{\contentsline {table}{\numberline {27}{\ignorespaces Bounded-tanh selector diagnostic over already measured candidates. The learned context+tangent selector is fit on validation tanh rows and evaluated once on held-out test tanh rows. It is action-bound-clean, but not a train-clean deployment selector.}}{32}{table.27}\protected@file@percent }
|
| 122 |
+
\newlabel{tab:ctt-base-context-obs-tanh-learned-dominance}{{27}{32}{Bounded-tanh selector diagnostic over already measured candidates. The learned context+tangent selector is fit on validation tanh rows and evaluated once on held-out test tanh rows. It is action-bound-clean, but not a train-clean deployment selector}{table.27}{}}
|
| 123 |
+
\@writefile{lot}{\contentsline {table}{\numberline {28}{\ignorespaces Train-calibrated bounded-tanh selector diagnostic. Calibration uses train-split tanh measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test tanh rows.}}{32}{table.28}\protected@file@percent }
|
| 124 |
+
\newlabel{tab:ctt-base-context-obs-tanh-learned-train-dominance}{{28}{32}{Train-calibrated bounded-tanh selector diagnostic. Calibration uses train-split tanh measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test tanh rows}{table.28}{}}
|
| 125 |
+
\@writefile{lot}{\contentsline {table}{\numberline {29}{\ignorespaces Train-calibrated \texttt {env\_clip} selector diagnostic. Calibration uses train-split \texttt {env\_clip} measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test \texttt {env\_clip} rows.}}{32}{table.29}\protected@file@percent }
|
| 126 |
+
\newlabel{tab:ctt-base-context-obs-envclip-learned-train-dominance}{{29}{32}{Train-calibrated \texttt {env\_clip} selector diagnostic. Calibration uses train-split \texttt {env\_clip} measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test \texttt {env\_clip} rows}{table.29}{}}
|
| 127 |
+
\@writefile{lot}{\contentsline {table}{\numberline {30}{\ignorespaces Train-calibrated \texttt {env\_clip} source-evidence selector diagnostic. This selector may read train-only source-chart positive/negative tangent statistics because \textsc {CTT}{} proposals are transported from measured train positive source tangents; it does not read validation/test outcomes.}}{32}{table.30}\protected@file@percent }
|
| 128 |
+
\newlabel{tab:ctt-base-context-obs-envclip-source-evidence-dominance}{{30}{32}{Train-calibrated \texttt {env\_clip} source-evidence selector diagnostic. This selector may read train-only source-chart positive/negative tangent statistics because \ctt {} proposals are transported from measured train positive source tangents; it does not read validation/test outcomes}{table.30}{}}
|
| 129 |
+
\@writefile{lot}{\contentsline {table}{\numberline {31}{\ignorespaces Held-out test \texttt {env\_clip} measured rollout at K=16. The table is generated by \texttt {scripts/eval\_metrics.py}; candidates are actually rolled out, so \ensuremath {\mathrm {OutcomePTR}}{}, SupportGap, and SelectorRegret are measured rather than proxy quantities.}}{33}{table.31}\protected@file@percent }
|
| 130 |
+
\newlabel{tab:ctt-base-context-obs-envclip-k16-test-rollout}{{31}{33}{Held-out test \texttt {env\_clip} measured rollout at K=16. The table is generated by \texttt {scripts/eval\_metrics.py}; candidates are actually rolled out, so \outcomeptr {}, SupportGap, and SelectorRegret are measured rather than proxy quantities}{table.31}{}}
|
| 131 |
+
\@writefile{lot}{\contentsline {table}{\numberline {32}{\ignorespaces Train-calibrated lower-confidence dominance fallback on the same held-out K=16 \texttt {env\_clip} measured rows. The conformal residual quantile and threshold are fit on train-calibration rows only. The artifact now reports action-bound unsafe execution and within-chart pairwise causal calibration error.}}{33}{table.32}\protected@file@percent }
|
| 132 |
+
\newlabel{tab:ctt-base-context-obs-envclip-k16-lcb-dominance}{{32}{33}{Train-calibrated lower-confidence dominance fallback on the same held-out K=16 \texttt {env\_clip} measured rows. The conformal residual quantile and threshold are fit on train-calibration rows only. The artifact now reports action-bound unsafe execution and within-chart pairwise causal calibration error}{table.32}{}}
|
| 133 |
+
\@writefile{lot}{\contentsline {table}{\numberline {33}{\ignorespaces Best current train-calibrated \texttt {env\_clip} K=16 selector diagnostic. Calibration uses only train-split K=16 measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test K=16 rows.}}{33}{table.33}\protected@file@percent }
|
| 134 |
+
\newlabel{tab:ctt-base-context-obs-envclip-k16-learned-train-dominance}{{33}{33}{Best current train-calibrated \texttt {env\_clip} K=16 selector diagnostic. Calibration uses only train-split K=16 measured rollout rows with same-chart and same-state source retrieval excluded, then evaluates once on held-out test K=16 rows}{table.33}{}}
|
| 135 |
+
\@writefile{lot}{\contentsline {table}{\numberline {34}{\ignorespaces Selector diagnostic sweep summary generated from completed selector run artifacts. For each action convention, the table reports the best row by held-out selected success while retaining all candidate rows in the \texttt {metrics.json} artifact. The comparison prevents choosing the bounded \texttt {tanh} selected-success row as the main result when its proposal support is much lower than K=16 \texttt {env\_clip}.}}{34}{table.34}\protected@file@percent }
|
| 136 |
+
\newlabel{tab:ctt-selector-diagnostic-sweep}{{34}{34}{Selector diagnostic sweep summary generated from completed selector run artifacts. For each action convention, the table reports the best row by held-out selected success while retaining all candidate rows in the \texttt {metrics.json} artifact. The comparison prevents choosing the bounded \texttt {tanh} selected-success row as the main result when its proposal support is much lower than K=16 \texttt {env\_clip}}{table.34}{}}
|
| 137 |
+
\@writefile{lot}{\contentsline {table}{\numberline {35}{\ignorespaces Measured outcome acceptance gate for the current K=16 \texttt {env\_clip} CTT result. The gate combines the measured rollout support artifact with the best train-clean K=16 selector and makes explicit which Part-F bars are still unmet.}}{34}{table.35}\protected@file@percent }
|
| 138 |
+
\newlabel{tab:ctt-outcome-acceptance-gate}{{35}{34}{Measured outcome acceptance gate for the current K=16 \texttt {env\_clip} CTT result. The gate combines the measured rollout support artifact with the best train-clean K=16 selector and makes explicit which Part-F bars are still unmet}{table.35}{}}
|
| 139 |
+
\@writefile{lot}{\contentsline {table}{\numberline {36}{\ignorespaces Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate.}}{34}{table.36}\protected@file@percent }
|
| 140 |
+
\newlabel{tab:ctt-base-context-obs-learned-train-dominance}{{36}{34}{Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate}{table.36}{}}
|
| 141 |
+
\@writefile{lot}{\contentsline {table}{\numberline {37}{\ignorespaces Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table\nobreakspace {}\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector.}}{34}{table.37}\protected@file@percent }
|
| 142 |
+
\newlabel{tab:ctt-base-context-obs-nonlinear-train-dominance}{{37}{34}{Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table~\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector}{table.37}{}}
|
| 143 |
+
\@writefile{lot}{\contentsline {table}{\numberline {38}{\ignorespaces Claim-to-artifact audit for this draft. The audit is generated by \texttt {scripts/audit\_ctt\_paper\_artifacts.py}. Warnings track the advisor's full run-contract fields such as per-run Markdown reports and logs; the current workspace policy keeps persistent prose consolidated in \texttt {README.md}.}}{34}{table.38}\protected@file@percent }
|
| 144 |
+
\newlabel{tab:paper-ctt-artifact-audit}{{38}{34}{Claim-to-artifact audit for this draft. The audit is generated by \texttt {scripts/audit\_ctt\_paper\_artifacts.py}. Warnings track the advisor's full run-contract fields such as per-run Markdown reports and logs; the current workspace policy keeps persistent prose consolidated in \texttt {README.md}}{table.38}{}}
|
| 145 |
+
\gdef \@abspage@last{34}
|
workspace/latex/main.fdb_latexmk
CHANGED
|
@@ -1,20 +1,20 @@
|
|
| 1 |
# Fdb version 4
|
| 2 |
-
["bibtex main"]
|
| 3 |
"./references.bib" 1783037215 3572 027cd2cf27b344368db784320d235a8c ""
|
| 4 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/bibtex/bst/base/plain.bst" 1292289607 20613 bd3fbfa9f64872b81ac57a0dd2ed855f ""
|
| 5 |
-
"main.aux"
|
| 6 |
(generated)
|
| 7 |
"main.bbl"
|
| 8 |
"main.blg"
|
| 9 |
(rewritten before read)
|
| 10 |
-
["pdflatex"]
|
| 11 |
"../paper/sections/theory.tex" 1783037322 4023 e8c0485606fdfea0020faae044bf1189 ""
|
| 12 |
"../runs/action_bound_audit_rgb_refs/table.tex" 1783102548 379 5a7d42a0f3c58a06c4eeeaa280615fb3 ""
|
| 13 |
"../runs/ctt_base_context_obj_val_rollout_comparison/table.tex" 1783103591 2172 16a82c2b9b79cbd8d2afdfca5503a296 ""
|
| 14 |
"../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex" 1783131816 380 bd815bba39d3063468fbda949042ce2e ""
|
| 15 |
"../runs/ctt_base_context_obs_learned_dominance_basic_envclip_train_to_test/table.tex" 1783121047 342 4ae8602b8d1e8ca3eb0baa82d0311b8d ""
|
| 16 |
"../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex" 1783130461 363 cb90b6f3f23ea450e770d398b0ab1413 ""
|
| 17 |
-
"../runs/ctt_base_context_obs_learned_dominance_context_success_tanh_train_to_test/table.tex"
|
| 18 |
"../runs/ctt_base_context_obs_learned_dominance_context_tangent_success_tanh_val_to_test/table.tex" 1783105123 342 dd9a1e5b4a221f68ef356cf5d2587225 ""
|
| 19 |
"../runs/ctt_base_context_obs_learned_dominance_context_val_to_test/table.tex" 1783094939 342 289b441acb02d293402c3de06a7e2f7a ""
|
| 20 |
"../runs/ctt_base_context_obs_learned_dominance_source_envclip_train_to_test/table.tex" 1783122702 342 289ce6968fc4a468108ffcc3600b954f ""
|
|
@@ -23,13 +23,13 @@
|
|
| 23 |
"../runs/ctt_base_context_obs_test_envclip_k16_rollout_comparison/table.tex" 1783123582 2544 3e7bc605cc803710efd73a65f6ae9c2c ""
|
| 24 |
"../runs/ctt_base_context_obs_test_envclip_rollout_comparison/table.tex" 1783120980 2516 76a70f818600a600a645ca511f8631a0 ""
|
| 25 |
"../runs/ctt_base_context_obs_test_noclip_rollout_comparison/table.tex" 1783103591 2514 53eb0178d1f721f8d1d39db3b88e715d ""
|
| 26 |
-
"../runs/ctt_base_context_obs_test_perdim_trainmax_rollout_comparison/table.tex"
|
| 27 |
"../runs/ctt_base_context_obs_test_rollout_comparison/table.tex" 1783103578 2170 0ea3761f3f17352fa51b5f8b4979fb70 ""
|
| 28 |
"../runs/ctt_base_context_obs_test_scaled0215_rollout_comparison/table.tex" 1783103317 2523 835503f4c8c1c02ad3e6761d733eef19 ""
|
| 29 |
"../runs/ctt_base_context_obs_test_tanh_rollout_comparison/table.tex" 1783104884 2518 7c061ede8cc8e34656555253887a5e0c ""
|
| 30 |
"../runs/ctt_base_context_obs_val_envclip_rollout_comparison/table.tex" 1783120978 2516 59fd79135a4ada5d2783b6937fb1b494 ""
|
| 31 |
"../runs/ctt_base_context_obs_val_noclip_rollout_comparison/table.tex" 1783103592 2516 efab11865e5c18448639a1adb45fc9f7 ""
|
| 32 |
-
"../runs/ctt_base_context_obs_val_perdim_trainmax_rollout_comparison/table.tex"
|
| 33 |
"../runs/ctt_base_context_obs_val_rollout_comparison/table.tex" 1783103576 2170 ada38b0197c9b24996df6ca37ebf8ce6 ""
|
| 34 |
"../runs/ctt_base_context_obs_val_scaled0215_rollout_comparison/table.tex" 1783103295 2522 dfa68ab52072ac5bcb2d63907e5bf8ca ""
|
| 35 |
"../runs/ctt_base_context_obs_val_tanh_rollout_comparison/table.tex" 1783104884 2517 db488fba06c06ee1ffcf630017eeef41 ""
|
|
@@ -37,11 +37,12 @@
|
|
| 37 |
"../runs/ctt_learned_dominance_val_to_test/table.tex" 1783051115 342 1951fdb40cb81709d46ad2391233a02b ""
|
| 38 |
"../runs/ctt_outcome_acceptance_gate/table.tex" 1783138334 749 116491f873dcdfa3ed98288748e0c899 ""
|
| 39 |
"../runs/ctt_residual_smoke_proxy/table.tex" 1783037068 479 a3687dfb2376b4f4dbfc05bec09c73c2 ""
|
|
|
|
| 40 |
"../runs/ctt_test_rollout_comparison/table.tex" 1783103576 2171 9cc76279db5d4ffeb30bd3108263eda6 ""
|
| 41 |
"../runs/ctt_val_proxy_comparison/table.tex" 1783137797 1201 46b7a5babb84523c9075eeafcf5c151c ""
|
| 42 |
"../runs/ctt_val_rollout_comparison/table.tex" 1783103576 1874 6e2344e80436f3a5295e8d9c128d776d ""
|
| 43 |
"../runs/data_accounting/table.tex" 1783034968 472 a35c3885ec2fd2c61f296ca898e5ee0e ""
|
| 44 |
-
"../runs/paper_ctt_audit/table.tex"
|
| 45 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-ts1.enc" 1136849721 2900 1537cc8184ad1792082cd229ecc269f4 ""
|
| 46 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/map/fontname/texfonts.map" 1577235249 3524 cb3e574dea2d1052e39280babc910dc8 ""
|
| 47 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/tfm/jknappen/ec/tcrm1000.tfm" 1136768653 1536 e07581a4bb3136ece9eeb4c3ffab8233 ""
|
|
@@ -161,10 +162,10 @@
|
|
| 161 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/web2c/texmf.cnf" 1692823885 39639 818dd8e5a0dce3a72597a7e6bad45d55 ""
|
| 162 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map" 1692821535 5031293 07973d5e761f645996ac32ebf7b8d1f3 ""
|
| 163 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/web2c/pdftex/pdflatex.fmt" 1692821530 1386016 abd9ba6eb53c47bc5de8d672b6ac87d3 ""
|
| 164 |
-
"main.aux"
|
| 165 |
-
"main.bbl"
|
| 166 |
-
"main.out"
|
| 167 |
-
"main.tex"
|
| 168 |
"tables/car_decomposition.tex" 1783032219 662 bac24c0014c2ba573860ce199b776934 ""
|
| 169 |
"tables/main_results.tex" 1783004655 1061 19bfdcc844614aadefc921dae15c700d ""
|
| 170 |
"tables/selector_calibration.tex" 1783011198 788 1b54a8eb88519c50862f4acf718798ab ""
|
|
|
|
| 1 |
# Fdb version 4
|
| 2 |
+
["bibtex main"] 1783140486 "main.aux" "main.bbl" "main" 1783140515 0
|
| 3 |
"./references.bib" 1783037215 3572 027cd2cf27b344368db784320d235a8c ""
|
| 4 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/bibtex/bst/base/plain.bst" 1292289607 20613 bd3fbfa9f64872b81ac57a0dd2ed855f ""
|
| 5 |
+
"main.aux" 1783140515 30971 dbae25d82dde9b0ed32296120ac93190 "pdflatex"
|
| 6 |
(generated)
|
| 7 |
"main.bbl"
|
| 8 |
"main.blg"
|
| 9 |
(rewritten before read)
|
| 10 |
+
["pdflatex"] 1783140514 "main.tex" "main.pdf" "main" 1783140515 0
|
| 11 |
"../paper/sections/theory.tex" 1783037322 4023 e8c0485606fdfea0020faae044bf1189 ""
|
| 12 |
"../runs/action_bound_audit_rgb_refs/table.tex" 1783102548 379 5a7d42a0f3c58a06c4eeeaa280615fb3 ""
|
| 13 |
"../runs/ctt_base_context_obj_val_rollout_comparison/table.tex" 1783103591 2172 16a82c2b9b79cbd8d2afdfca5503a296 ""
|
| 14 |
"../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex" 1783131816 380 bd815bba39d3063468fbda949042ce2e ""
|
| 15 |
"../runs/ctt_base_context_obs_learned_dominance_basic_envclip_train_to_test/table.tex" 1783121047 342 4ae8602b8d1e8ca3eb0baa82d0311b8d ""
|
| 16 |
"../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex" 1783130461 363 cb90b6f3f23ea450e770d398b0ab1413 ""
|
| 17 |
+
"../runs/ctt_base_context_obs_learned_dominance_context_success_tanh_train_to_test/table.tex" 1783139605 363 dc8396ddd5f4b542770e44d200bc5ed0 ""
|
| 18 |
"../runs/ctt_base_context_obs_learned_dominance_context_tangent_success_tanh_val_to_test/table.tex" 1783105123 342 dd9a1e5b4a221f68ef356cf5d2587225 ""
|
| 19 |
"../runs/ctt_base_context_obs_learned_dominance_context_val_to_test/table.tex" 1783094939 342 289b441acb02d293402c3de06a7e2f7a ""
|
| 20 |
"../runs/ctt_base_context_obs_learned_dominance_source_envclip_train_to_test/table.tex" 1783122702 342 289ce6968fc4a468108ffcc3600b954f ""
|
|
|
|
| 23 |
"../runs/ctt_base_context_obs_test_envclip_k16_rollout_comparison/table.tex" 1783123582 2544 3e7bc605cc803710efd73a65f6ae9c2c ""
|
| 24 |
"../runs/ctt_base_context_obs_test_envclip_rollout_comparison/table.tex" 1783120980 2516 76a70f818600a600a645ca511f8631a0 ""
|
| 25 |
"../runs/ctt_base_context_obs_test_noclip_rollout_comparison/table.tex" 1783103591 2514 53eb0178d1f721f8d1d39db3b88e715d ""
|
| 26 |
+
"../runs/ctt_base_context_obs_test_perdim_trainmax_rollout_comparison/table.tex" 1783139436 2523 901a0e931d1b515e9dc956b081b93b78 ""
|
| 27 |
"../runs/ctt_base_context_obs_test_rollout_comparison/table.tex" 1783103578 2170 0ea3761f3f17352fa51b5f8b4979fb70 ""
|
| 28 |
"../runs/ctt_base_context_obs_test_scaled0215_rollout_comparison/table.tex" 1783103317 2523 835503f4c8c1c02ad3e6761d733eef19 ""
|
| 29 |
"../runs/ctt_base_context_obs_test_tanh_rollout_comparison/table.tex" 1783104884 2518 7c061ede8cc8e34656555253887a5e0c ""
|
| 30 |
"../runs/ctt_base_context_obs_val_envclip_rollout_comparison/table.tex" 1783120978 2516 59fd79135a4ada5d2783b6937fb1b494 ""
|
| 31 |
"../runs/ctt_base_context_obs_val_noclip_rollout_comparison/table.tex" 1783103592 2516 efab11865e5c18448639a1adb45fc9f7 ""
|
| 32 |
+
"../runs/ctt_base_context_obs_val_perdim_trainmax_rollout_comparison/table.tex" 1783139434 2523 ab8f8d646407958b1d62fa9cac540d97 ""
|
| 33 |
"../runs/ctt_base_context_obs_val_rollout_comparison/table.tex" 1783103576 2170 ada38b0197c9b24996df6ca37ebf8ce6 ""
|
| 34 |
"../runs/ctt_base_context_obs_val_scaled0215_rollout_comparison/table.tex" 1783103295 2522 dfa68ab52072ac5bcb2d63907e5bf8ca ""
|
| 35 |
"../runs/ctt_base_context_obs_val_tanh_rollout_comparison/table.tex" 1783104884 2517 db488fba06c06ee1ffcf630017eeef41 ""
|
|
|
|
| 37 |
"../runs/ctt_learned_dominance_val_to_test/table.tex" 1783051115 342 1951fdb40cb81709d46ad2391233a02b ""
|
| 38 |
"../runs/ctt_outcome_acceptance_gate/table.tex" 1783138334 749 116491f873dcdfa3ed98288748e0c899 ""
|
| 39 |
"../runs/ctt_residual_smoke_proxy/table.tex" 1783037068 479 a3687dfb2376b4f4dbfc05bec09c73c2 ""
|
| 40 |
+
"../runs/ctt_selector_diagnostic_sweep/table.tex" 1783140431 553 95e3be80df669abd25cfc0552a6d7261 ""
|
| 41 |
"../runs/ctt_test_rollout_comparison/table.tex" 1783103576 2171 9cc76279db5d4ffeb30bd3108263eda6 ""
|
| 42 |
"../runs/ctt_val_proxy_comparison/table.tex" 1783137797 1201 46b7a5babb84523c9075eeafcf5c151c ""
|
| 43 |
"../runs/ctt_val_rollout_comparison/table.tex" 1783103576 1874 6e2344e80436f3a5295e8d9c128d776d ""
|
| 44 |
"../runs/data_accounting/table.tex" 1783034968 472 a35c3885ec2fd2c61f296ca898e5ee0e ""
|
| 45 |
+
"../runs/paper_ctt_audit/table.tex" 1783140507 332 bbc7a0fd63377a4ed4382f5d91787f3e ""
|
| 46 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/enc/dvips/cm-super/cm-super-ts1.enc" 1136849721 2900 1537cc8184ad1792082cd229ecc269f4 ""
|
| 47 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/map/fontname/texfonts.map" 1577235249 3524 cb3e574dea2d1052e39280babc910dc8 ""
|
| 48 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/tfm/jknappen/ec/tcrm1000.tfm" 1136768653 1536 e07581a4bb3136ece9eeb4c3ffab8233 ""
|
|
|
|
| 162 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/web2c/texmf.cnf" 1692823885 39639 818dd8e5a0dce3a72597a7e6bad45d55 ""
|
| 163 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/fonts/map/pdftex/updmap/pdftex.map" 1692821535 5031293 07973d5e761f645996ac32ebf7b8d1f3 ""
|
| 164 |
"/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/var/lib/texmf/web2c/pdftex/pdflatex.fmt" 1692821530 1386016 abd9ba6eb53c47bc5de8d672b6ac87d3 ""
|
| 165 |
+
"main.aux" 1783140515 30971 dbae25d82dde9b0ed32296120ac93190 "pdflatex"
|
| 166 |
+
"main.bbl" 1783140486 3120 92555a174bae4ac684748bbe3fae0b8c "bibtex main"
|
| 167 |
+
"main.out" 1783140515 2647 6c337a545f287573528c7cb52648bc95 "pdflatex"
|
| 168 |
+
"main.tex" 1783140456 68989 39c3e3573f9ebe9746abf7207609b1c2 ""
|
| 169 |
"tables/car_decomposition.tex" 1783032219 662 bac24c0014c2ba573860ce199b776934 ""
|
| 170 |
"tables/main_results.tex" 1783004655 1061 19bfdcc844614aadefc921dae15c700d ""
|
| 171 |
"tables/selector_calibration.tex" 1783011198 788 1b54a8eb88519c50862f4acf718798ab ""
|
workspace/latex/main.fls
CHANGED
|
@@ -999,6 +999,17 @@ INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_tas
|
|
| 999 |
INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
|
| 1000 |
INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
|
| 1001 |
INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1002 |
INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
|
| 1003 |
INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
|
| 1004 |
INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
|
|
|
|
| 999 |
INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
|
| 1000 |
INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
|
| 1001 |
INPUT ../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_envclip_k16_train_to_test/table.tex
|
| 1002 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1003 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1004 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1005 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1006 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1007 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1008 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1009 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1010 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1011 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1012 |
+
INPUT ../runs/ctt_selector_diagnostic_sweep/table.tex
|
| 1013 |
INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
|
| 1014 |
INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
|
| 1015 |
INPUT ../runs/ctt_outcome_acceptance_gate/table.tex
|
workspace/latex/main.log
CHANGED
|
@@ -1,4 +1,4 @@
|
|
| 1 |
-
This is pdfTeX, Version 3.141592653-2.6-1.40.22 (TeX Live 2021 Gentoo Linux) (preloaded format=pdflatex 2023.8.23) 4 JUL 2026 00:
|
| 2 |
entering extended mode
|
| 3 |
restricted \write18 enabled.
|
| 4 |
%&-line parsing enabled.
|
|
@@ -602,119 +602,119 @@ Underfull \hbox (badness 3439) in paragraph at lines 972--972
|
|
| 602 |
(../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex)
|
| 603 |
(../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_en
|
| 604 |
vclip_k16_train_to_test/table.tex)
|
| 605 |
-
(../runs/
|
|
|
|
| 606 |
(../runs/ctt_base_context_obs_learned_dominance_train_to_test/table.tex)
|
| 607 |
-
[12]
|
| 608 |
(../runs/ctt_base_context_obs_nonlinear_dominance_basic_positive_train_to_test/
|
| 609 |
table.tex)
|
| 610 |
-
Overfull \hbox (31.94885pt too wide) in paragraph at lines
|
| 611 |
[]\OT1/cmtt/m/n/10 scripts/export[]cil[]charts.py\OT1/cmr/m/n/10 , \OT1/cmtt/m/
|
| 612 |
n/10 scripts/build[]data[]accounting.py\OT1/cmr/m/n/10 , and \OT1/cmtt/m/n/10 s
|
| 613 |
cripts/audit[]cil[]charts.py
|
| 614 |
[]
|
| 615 |
|
| 616 |
|
| 617 |
-
Overfull \hbox (127.11275pt too wide) in paragraph at lines
|
| 618 |
[]\OT1/cmtt/m/n/10 cil/chart[]features.py \OT1/cmr/m/n/10 cen-tral-izes deploym
|
| 619 |
ent-visible chart fea-ture con-struc-tion, and \OT1/cmtt/m/n/10 scripts/audit[]
|
| 620 |
chart[]feature[]sources.py
|
| 621 |
[]
|
| 622 |
|
| 623 |
|
| 624 |
-
Overfull \hbox (104.78587pt too wide) in paragraph at lines
|
| 625 |
[]\OT1/cmtt/m/n/10 scripts/slurm/render[]six[]task[]chart[]observations.sbatch
|
| 626 |
\OT1/cmr/m/n/10 and \OT1/cmtt/m/n/10 scripts/slurm/reexport[]rgb[]ref[]cil[]cha
|
| 627 |
rts.sbatch
|
| 628 |
[]
|
| 629 |
|
| 630 |
|
| 631 |
-
Overfull \hbox (23.72661pt too wide) in paragraph at lines
|
| 632 |
[]\OT1/cmtt/m/n/10 scripts/export[]chart[]observation[]embeddings.py \OT1/cmr/m
|
| 633 |
/n/10 and \OT1/cmtt/m/n/10 scripts/export[]chart[]object[]embeddings.py
|
| 634 |
[]
|
| 635 |
|
| 636 |
|
| 637 |
-
Overfull \hbox (129.91359pt too wide) in paragraph at lines
|
| 638 |
\OT1/cmr/m/n/10 cre-ate de-ter-min-is-tic 32D RGB-stat and 64D RGB object-layou
|
| 639 |
t em-bed-dings, and \OT1/cmtt/m/n/10 scripts/slurm/train[]ctt[]feature[]proxy.s
|
| 640 |
batch
|
| 641 |
[]
|
| 642 |
|
| 643 |
|
| 644 |
-
Overfull \hbox (7.97675pt too wide) in paragraph at lines
|
| 645 |
[]\OT1/cmtt/m/n/10 scripts/eval[]ctt[]generated[]rollout.py \OT1/cmr/m/n/10 and
|
| 646 |
\OT1/cmtt/m/n/10 scripts/slurm/eval[]ctt[]generated[]rollout.sbatch
|
| 647 |
[]
|
| 648 |
|
| 649 |
|
| 650 |
-
Overfull \hbox (20.61057pt too wide) in paragraph at lines
|
| 651 |
\OT1/cmr/m/n/10 train-split cal-i-bra-tion and meta-data load-ing for deploymen
|
| 652 |
t-visible chart fea-tures such as \OT1/cmtt/m/n/10 base[]context[]obs\OT1/cmr/m
|
| 653 |
/n/10 .
|
| 654 |
[]
|
| 655 |
|
| 656 |
|
| 657 |
-
Overfull \hbox (49.05014pt too wide) in paragraph at lines
|
| 658 |
\OT1/cmr/m/n/10 They also record action-bound va-lid-ity la-bels and sup-port r
|
| 659 |
aw-action re-play through \OT1/cmtt/m/n/10 --disable-action-clipping
|
| 660 |
[]
|
| 661 |
|
| 662 |
|
| 663 |
-
Overfull \hbox (2.58342pt too wide) in paragraph at lines
|
| 664 |
[]\OT1/cmtt/m/n/10 scripts/eval[]nonlinear[]dominance[]selector.py \OT1/cmr/m/n
|
| 665 |
/10 shares the chart-compatibility feature-loading path
|
| 666 |
[]
|
| 667 |
|
| 668 |
|
| 669 |
-
Overfull \hbox (11.7557pt too wide) in paragraph at lines
|
| 670 |
[]\OT1/cmtt/m/n/10 scripts/audit[]action[]bounds.py \OT1/cmr/m/n/10 pro-duces \
|
| 671 |
OT1/cmtt/m/n/10 runs/action[]bound[]audit[]rgb[]refs\OT1/cmr/m/n/10 , which au-
|
| 672 |
dits whether
|
| 673 |
[]
|
| 674 |
|
| 675 |
[13]
|
| 676 |
-
Overfull \hbox (6.0625pt too wide) in paragraph at lines
|
| 677 |
[]\OT1/cmtt/m/n/10 scripts/eval[]chart[]positive[]memory[]proxy.py \OT1/cmr/m/n
|
| 678 |
/10 and \OT1/cmtt/m/n/10 scripts/build[]ctt[]proxy[]comparison.py \OT1/cmr/m/n/
|
| 679 |
10 gen-
|
| 680 |
[]
|
| 681 |
|
| 682 |
|
| 683 |
-
Overfull \hbox (63.81772pt too wide) in paragraph at lines
|
| 684 |
\OT1/cmr/m/n/10 min-is-tic sum-maries of \OT1/cmtt/m/n/10 delta[]action\OT1/cmr
|
| 685 |
/m/n/10 ; \OT1/cmtt/m/n/10 runs/tangent[]reconstruction \OT1/cmr/m/n/10 and \OT
|
| 686 |
1/cmtt/m/n/10 runs/tangent[]reconstruction[]rgb[]refs
|
| 687 |
[]
|
| 688 |
|
| 689 |
|
| 690 |
-
Overfull \hbox (61.283pt too wide) in paragraph at lines
|
| 691 |
[]\OT1/cmtt/m/n/10 scripts/audit[]cil[]charts.py \OT1/cmr/m/n/10 writes the lea
|
| 692 |
k-age re-ports \OT1/cmtt/m/n/10 runs/leakage[]audit \OT1/cmr/m/n/10 and \OT1/cm
|
| 693 |
tt/m/n/10 runs/leakage[]audit[]rgb[]refs\OT1/cmr/m/n/10 ,
|
| 694 |
[]
|
| 695 |
|
| 696 |
|
| 697 |
-
Overfull \hbox (20.3325pt too wide) in paragraph at lines
|
| 698 |
[]\OT1/cmtt/m/n/10 scripts/train[]utility[]energy.py \OT1/cmr/m/n/10 and \OT1/c
|
| 699 |
mtt/m/n/10 scripts/calibrate[]dominance.py \OT1/cmr/m/n/10 im-ple-ment the util
|
| 700 |
-ity/scoring
|
| 701 |
[]
|
| 702 |
|
| 703 |
|
| 704 |
-
Overfull \hbox (3.63927pt too wide) in paragraph at lines
|
| 705 |
[]\OT1/cmtt/m/n/10 scripts/backfill[]paper[]run[]artifacts.py \OT1/cmr/m/n/10 t
|
| 706 |
rans-par-ently back-fills non-Markdown run meta-data such
|
| 707 |
[]
|
| 708 |
|
| 709 |
|
| 710 |
-
Underfull \hbox (badness 10000) in paragraph at lines
|
| 711 |
-
[]\OT1/cmr/m/n/10 Table
|
| 712 |
it is gen-er-ated by
|
| 713 |
[]
|
| 714 |
|
| 715 |
-
(../runs/paper_ctt_audit/table.tex) [14] (./main.bbl
|
| 716 |
[19] [20] [21] [22] [23] [24] [25] [26] [27] [28] [29] [30] [31] [32] [33]
|
| 717 |
-
(./main.aux)
|
| 718 |
|
| 719 |
LaTeX Font Warning: Some font shapes were not available, defaults substituted.
|
| 720 |
|
|
@@ -722,13 +722,13 @@ Package rerunfilecheck Info: File `main.out' has not changed.
|
|
| 722 |
(rerunfilecheck) Checksum: 6C337A545F287573528C7CB52648BC95;2647.
|
| 723 |
)
|
| 724 |
Here is how much of TeX's memory you used:
|
| 725 |
-
|
| 726 |
-
|
| 727 |
-
|
| 728 |
-
|
| 729 |
414764 words of font info for 71 fonts, out of 8000000 for 9000
|
| 730 |
36 hyphenation exceptions out of 8191
|
| 731 |
-
71i,10n,74p,608b,
|
| 732 |
{/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/font
|
| 733 |
s/enc/dvips/cm-super/cm-super-ts1.enc}</cvmfs/soft.computecanada.ca/gentoo/2023
|
| 734 |
/x86-64-v3/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmbx10.pfb></cvm
|
|
@@ -769,10 +769,10 @@ cm/cmtt9.pfb></cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texm
|
|
| 769 |
f-dist/fonts/type1/public/amsfonts/symbols/msbm10.pfb></cvmfs/soft.computecanad
|
| 770 |
a.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/type1/public/cm-super/sfr
|
| 771 |
m1000.pfb>
|
| 772 |
-
Output written on main.pdf (
|
| 773 |
PDF statistics:
|
| 774 |
-
|
| 775 |
-
|
| 776 |
-
|
| 777 |
137 words of extra memory for PDF output out of 10000 (max. 10000000)
|
| 778 |
|
|
|
|
| 1 |
+
This is pdfTeX, Version 3.141592653-2.6-1.40.22 (TeX Live 2021 Gentoo Linux) (preloaded format=pdflatex 2023.8.23) 4 JUL 2026 00:48
|
| 2 |
entering extended mode
|
| 3 |
restricted \write18 enabled.
|
| 4 |
%&-line parsing enabled.
|
|
|
|
| 602 |
(../runs/ctt_base_context_obs_dominance_envclip_k16_train_to_test/table.tex)
|
| 603 |
(../runs/ctt_base_context_obs_learned_dominance_chartcompat_obs_utility_task_en
|
| 604 |
vclip_k16_train_to_test/table.tex)
|
| 605 |
+
(../runs/ctt_selector_diagnostic_sweep/table.tex)
|
| 606 |
+
(../runs/ctt_outcome_acceptance_gate/table.tex) [12]
|
| 607 |
(../runs/ctt_base_context_obs_learned_dominance_train_to_test/table.tex)
|
|
|
|
| 608 |
(../runs/ctt_base_context_obs_nonlinear_dominance_basic_positive_train_to_test/
|
| 609 |
table.tex)
|
| 610 |
+
Overfull \hbox (31.94885pt too wide) in paragraph at lines 1162--1166
|
| 611 |
[]\OT1/cmtt/m/n/10 scripts/export[]cil[]charts.py\OT1/cmr/m/n/10 , \OT1/cmtt/m/
|
| 612 |
n/10 scripts/build[]data[]accounting.py\OT1/cmr/m/n/10 , and \OT1/cmtt/m/n/10 s
|
| 613 |
cripts/audit[]cil[]charts.py
|
| 614 |
[]
|
| 615 |
|
| 616 |
|
| 617 |
+
Overfull \hbox (127.11275pt too wide) in paragraph at lines 1168--1172
|
| 618 |
[]\OT1/cmtt/m/n/10 cil/chart[]features.py \OT1/cmr/m/n/10 cen-tral-izes deploym
|
| 619 |
ent-visible chart fea-ture con-struc-tion, and \OT1/cmtt/m/n/10 scripts/audit[]
|
| 620 |
chart[]feature[]sources.py
|
| 621 |
[]
|
| 622 |
|
| 623 |
|
| 624 |
+
Overfull \hbox (104.78587pt too wide) in paragraph at lines 1172--1175
|
| 625 |
[]\OT1/cmtt/m/n/10 scripts/slurm/render[]six[]task[]chart[]observations.sbatch
|
| 626 |
\OT1/cmr/m/n/10 and \OT1/cmtt/m/n/10 scripts/slurm/reexport[]rgb[]ref[]cil[]cha
|
| 627 |
rts.sbatch
|
| 628 |
[]
|
| 629 |
|
| 630 |
|
| 631 |
+
Overfull \hbox (23.72661pt too wide) in paragraph at lines 1175--1180
|
| 632 |
[]\OT1/cmtt/m/n/10 scripts/export[]chart[]observation[]embeddings.py \OT1/cmr/m
|
| 633 |
/n/10 and \OT1/cmtt/m/n/10 scripts/export[]chart[]object[]embeddings.py
|
| 634 |
[]
|
| 635 |
|
| 636 |
|
| 637 |
+
Overfull \hbox (129.91359pt too wide) in paragraph at lines 1175--1180
|
| 638 |
\OT1/cmr/m/n/10 cre-ate de-ter-min-is-tic 32D RGB-stat and 64D RGB object-layou
|
| 639 |
t em-bed-dings, and \OT1/cmtt/m/n/10 scripts/slurm/train[]ctt[]feature[]proxy.s
|
| 640 |
batch
|
| 641 |
[]
|
| 642 |
|
| 643 |
|
| 644 |
+
Overfull \hbox (7.97675pt too wide) in paragraph at lines 1180--1189
|
| 645 |
[]\OT1/cmtt/m/n/10 scripts/eval[]ctt[]generated[]rollout.py \OT1/cmr/m/n/10 and
|
| 646 |
\OT1/cmtt/m/n/10 scripts/slurm/eval[]ctt[]generated[]rollout.sbatch
|
| 647 |
[]
|
| 648 |
|
| 649 |
|
| 650 |
+
Overfull \hbox (20.61057pt too wide) in paragraph at lines 1180--1189
|
| 651 |
\OT1/cmr/m/n/10 train-split cal-i-bra-tion and meta-data load-ing for deploymen
|
| 652 |
t-visible chart fea-tures such as \OT1/cmtt/m/n/10 base[]context[]obs\OT1/cmr/m
|
| 653 |
/n/10 .
|
| 654 |
[]
|
| 655 |
|
| 656 |
|
| 657 |
+
Overfull \hbox (49.05014pt too wide) in paragraph at lines 1180--1189
|
| 658 |
\OT1/cmr/m/n/10 They also record action-bound va-lid-ity la-bels and sup-port r
|
| 659 |
aw-action re-play through \OT1/cmtt/m/n/10 --disable-action-clipping
|
| 660 |
[]
|
| 661 |
|
| 662 |
|
| 663 |
+
Overfull \hbox (2.58342pt too wide) in paragraph at lines 1189--1193
|
| 664 |
[]\OT1/cmtt/m/n/10 scripts/eval[]nonlinear[]dominance[]selector.py \OT1/cmr/m/n
|
| 665 |
/10 shares the chart-compatibility feature-loading path
|
| 666 |
[]
|
| 667 |
|
| 668 |
|
| 669 |
+
Overfull \hbox (11.7557pt too wide) in paragraph at lines 1193--1196
|
| 670 |
[]\OT1/cmtt/m/n/10 scripts/audit[]action[]bounds.py \OT1/cmr/m/n/10 pro-duces \
|
| 671 |
OT1/cmtt/m/n/10 runs/action[]bound[]audit[]rgb[]refs\OT1/cmr/m/n/10 , which au-
|
| 672 |
dits whether
|
| 673 |
[]
|
| 674 |
|
| 675 |
[13]
|
| 676 |
+
Overfull \hbox (6.0625pt too wide) in paragraph at lines 1215--1218
|
| 677 |
[]\OT1/cmtt/m/n/10 scripts/eval[]chart[]positive[]memory[]proxy.py \OT1/cmr/m/n
|
| 678 |
/10 and \OT1/cmtt/m/n/10 scripts/build[]ctt[]proxy[]comparison.py \OT1/cmr/m/n/
|
| 679 |
10 gen-
|
| 680 |
[]
|
| 681 |
|
| 682 |
|
| 683 |
+
Overfull \hbox (63.81772pt too wide) in paragraph at lines 1218--1223
|
| 684 |
\OT1/cmr/m/n/10 min-is-tic sum-maries of \OT1/cmtt/m/n/10 delta[]action\OT1/cmr
|
| 685 |
/m/n/10 ; \OT1/cmtt/m/n/10 runs/tangent[]reconstruction \OT1/cmr/m/n/10 and \OT
|
| 686 |
1/cmtt/m/n/10 runs/tangent[]reconstruction[]rgb[]refs
|
| 687 |
[]
|
| 688 |
|
| 689 |
|
| 690 |
+
Overfull \hbox (61.283pt too wide) in paragraph at lines 1223--1227
|
| 691 |
[]\OT1/cmtt/m/n/10 scripts/audit[]cil[]charts.py \OT1/cmr/m/n/10 writes the lea
|
| 692 |
k-age re-ports \OT1/cmtt/m/n/10 runs/leakage[]audit \OT1/cmr/m/n/10 and \OT1/cm
|
| 693 |
tt/m/n/10 runs/leakage[]audit[]rgb[]refs\OT1/cmr/m/n/10 ,
|
| 694 |
[]
|
| 695 |
|
| 696 |
|
| 697 |
+
Overfull \hbox (20.3325pt too wide) in paragraph at lines 1227--1230
|
| 698 |
[]\OT1/cmtt/m/n/10 scripts/train[]utility[]energy.py \OT1/cmr/m/n/10 and \OT1/c
|
| 699 |
mtt/m/n/10 scripts/calibrate[]dominance.py \OT1/cmr/m/n/10 im-ple-ment the util
|
| 700 |
-ity/scoring
|
| 701 |
[]
|
| 702 |
|
| 703 |
|
| 704 |
+
Overfull \hbox (3.63927pt too wide) in paragraph at lines 1236--1240
|
| 705 |
[]\OT1/cmtt/m/n/10 scripts/backfill[]paper[]run[]artifacts.py \OT1/cmr/m/n/10 t
|
| 706 |
rans-par-ently back-fills non-Markdown run meta-data such
|
| 707 |
[]
|
| 708 |
|
| 709 |
|
| 710 |
+
Underfull \hbox (badness 10000) in paragraph at lines 1248--1248
|
| 711 |
+
[]\OT1/cmr/m/n/10 Table 38: []Claim-to-artifact au-dit for this draft. The au-d
|
| 712 |
it is gen-er-ated by
|
| 713 |
[]
|
| 714 |
|
| 715 |
+
(../runs/paper_ctt_audit/table.tex) [14] (./main.bbl [15]) [16] [17] [18]
|
| 716 |
[19] [20] [21] [22] [23] [24] [25] [26] [27] [28] [29] [30] [31] [32] [33]
|
| 717 |
+
[34] (./main.aux)
|
| 718 |
|
| 719 |
LaTeX Font Warning: Some font shapes were not available, defaults substituted.
|
| 720 |
|
|
|
|
| 722 |
(rerunfilecheck) Checksum: 6C337A545F287573528C7CB52648BC95;2647.
|
| 723 |
)
|
| 724 |
Here is how much of TeX's memory you used:
|
| 725 |
+
10295 strings out of 480884
|
| 726 |
+
159358 string characters out of 5900692
|
| 727 |
+
727858 words of memory out of 5000000
|
| 728 |
+
27161 multiletter control sequences out of 15000+600000
|
| 729 |
414764 words of font info for 71 fonts, out of 8000000 for 9000
|
| 730 |
36 hyphenation exceptions out of 8191
|
| 731 |
+
71i,10n,74p,608b,446s stack positions out of 5000i,500n,10000p,200000b,80000s
|
| 732 |
{/cvmfs/soft.computecanada.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/font
|
| 733 |
s/enc/dvips/cm-super/cm-super-ts1.enc}</cvmfs/soft.computecanada.ca/gentoo/2023
|
| 734 |
/x86-64-v3/usr/share/texmf-dist/fonts/type1/public/amsfonts/cm/cmbx10.pfb></cvm
|
|
|
|
| 769 |
f-dist/fonts/type1/public/amsfonts/symbols/msbm10.pfb></cvmfs/soft.computecanad
|
| 770 |
a.ca/gentoo/2023/x86-64-v3/usr/share/texmf-dist/fonts/type1/public/cm-super/sfr
|
| 771 |
m1000.pfb>
|
| 772 |
+
Output written on main.pdf (34 pages, 438012 bytes).
|
| 773 |
PDF statistics:
|
| 774 |
+
460 PDF objects out of 1000 (max. 8388607)
|
| 775 |
+
393 compressed objects within 4 object streams
|
| 776 |
+
118 named destinations out of 1000 (max. 500000)
|
| 777 |
137 words of extra memory for PDF output out of 10000 (max. 10000000)
|
| 778 |
|
workspace/latex/main.pdf
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cd3e913de31fb2f4a8c4a4c9393216a45036075bb97cc953339a6c61400872af
|
| 3 |
+
size 438012
|
workspace/latex/main.tex
CHANGED
|
@@ -1054,6 +1054,30 @@ conclusion is sharper than the K=8 result: support is now strong enough to make
|
|
| 1054 |
the paper story credible, train-only visual chart compatibility is useful, but
|
| 1055 |
selector/utility energy is still the dominant remaining bottleneck.
|
| 1056 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1057 |
\begin{table}[t]
|
| 1058 |
\centering
|
| 1059 |
\caption{Measured outcome acceptance gate for the current K=16
|
|
|
|
| 1054 |
the paper story credible, train-only visual chart compatibility is useful, but
|
| 1055 |
selector/utility energy is still the dominant remaining bottleneck.
|
| 1056 |
|
| 1057 |
+
\begin{table}[t]
|
| 1058 |
+
\centering
|
| 1059 |
+
\caption{Selector diagnostic sweep summary generated from completed selector
|
| 1060 |
+
run artifacts. For each action convention, the table reports the best row by
|
| 1061 |
+
held-out selected success while retaining all candidate rows in the
|
| 1062 |
+
\texttt{metrics.json} artifact. The comparison prevents choosing the bounded
|
| 1063 |
+
\texttt{tanh} selected-success row as the main result when its proposal support
|
| 1064 |
+
is much lower than K=16 \texttt{env\_clip}.}
|
| 1065 |
+
\label{tab:ctt-selector-diagnostic-sweep}
|
| 1066 |
+
\scriptsize
|
| 1067 |
+
\resizebox{\linewidth}{!}{\input{../runs/ctt_selector_diagnostic_sweep/table}}
|
| 1068 |
+
\end{table}
|
| 1069 |
+
|
| 1070 |
+
Table~\ref{tab:ctt-selector-diagnostic-sweep} summarizes the selector
|
| 1071 |
+
diagnostics without hiding the tradeoff. The K=8 bounded-\texttt{tanh} selector
|
| 1072 |
+
reaches the highest selected success, 38.19\%, but its proposal oracle is also
|
| 1073 |
+
only 38.19\%, so it is not a stronger support result. The K=8 per-dimension
|
| 1074 |
+
train-max convention mostly falls back to the base action and has still lower
|
| 1075 |
+
generated support. K=16 \texttt{env\_clip} remains the strongest support
|
| 1076 |
+
setting, with a 56.94\% proposal oracle and 54.86\% \outcomeptr{}@16, but it
|
| 1077 |
+
leaves the largest selector gap. Thus the honest conclusion is not that one
|
| 1078 |
+
post-processing convention solves deployment; it is that support and selection
|
| 1079 |
+
must both pass their gates.
|
| 1080 |
+
|
| 1081 |
\begin{table}[t]
|
| 1082 |
\centering
|
| 1083 |
\caption{Measured outcome acceptance gate for the current K=16
|