auto-sync 2026-07-03T16:53:36Z workspace
Browse files- workspace/latex/main.aux +14 -10
workspace/latex/main.aux
CHANGED
|
@@ -64,31 +64,35 @@
|
|
| 64 |
\newlabel{tab:ctt-val-proxy}{{7}{8}{Validation proxy comparison on 69 validation charts with measured positive tangents, using train-only source positives. The gate column is proxy-only: it requires no more than one point higher NegativeNear@0.20 than local-atlas and improvement on \pptc {}@0.20, \pptc {}@0.40, or mean positive distance. Passing this gate permits rollout evaluation; it is not \outcomeptr {} or measured success}{table.7}{}}
|
| 65 |
\@writefile{lot}{\contentsline {table}{\numberline {8}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{9}{table.8}\protected@file@percent }
|
| 66 |
\newlabel{tab:ctt-val-rollout}{{8}{9}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.8}{}}
|
| 67 |
-
\@writefile{toc}{\contentsline {section}{\numberline {8}Reproducibility Artifacts}{9}{section.8}\protected@file@percent }
|
| 68 |
\@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{10}{table.9}\protected@file@percent }
|
| 69 |
\newlabel{tab:ctt-base-context-obs-val-rollout}{{9}{10}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.9}{}}
|
|
|
|
| 70 |
\@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{11}{table.10}\protected@file@percent }
|
| 71 |
\newlabel{tab:ctt-test-rollout}{{10}{11}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.10}{}}
|
| 72 |
-
\@writefile{toc}{\contentsline {section}{\numberline {9}Limitations and Next Steps}{11}{section.9}\protected@file@percent }
|
| 73 |
-
\bibstyle{plain}
|
| 74 |
-
\bibdata{references}
|
| 75 |
-
\bibcite{chen2025robotwin2}{1}
|
| 76 |
-
\bibcite{glossop2025cast}{2}
|
| 77 |
\@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{12}{table.11}\protected@file@percent }
|
| 78 |
\newlabel{tab:ctt-base-context-obs-test-rollout}{{11}{12}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.11}{}}
|
| 79 |
\@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{12}{table.12}\protected@file@percent }
|
| 80 |
\newlabel{tab:ctt-dominance}{{12}{12}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.12}{}}
|
| 81 |
-
\@writefile{toc}{\contentsline {section}{\numberline {
|
|
|
|
|
|
|
|
|
|
|
|
|
| 82 |
\bibcite{kim2025openvlaoft}{3}
|
| 83 |
\bibcite{kim2024openvla}{4}
|
| 84 |
\bibcite{kwok2025robomonkey}{5}
|
| 85 |
\bibcite{liu2023libero}{6}
|
| 86 |
\bibcite{singh2026bokbo}{7}
|
| 87 |
\bibcite{tao2024maniskill3}{8}
|
| 88 |
-
\bibcite{zhang2024vlabench}{9}
|
| 89 |
-
\bibcite{zhao2026verispace}{10}
|
| 90 |
\@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate.}}{13}{table.13}\protected@file@percent }
|
| 91 |
\newlabel{tab:ctt-learned-dominance}{{13}{13}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate}{table.13}{}}
|
| 92 |
\@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{13}{table.14}\protected@file@percent }
|
| 93 |
\newlabel{tab:ctt-base-context-obs-learned-dominance}{{14}{13}{Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.14}{}}
|
| 94 |
-
\
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 64 |
\newlabel{tab:ctt-val-proxy}{{7}{8}{Validation proxy comparison on 69 validation charts with measured positive tangents, using train-only source positives. The gate column is proxy-only: it requires no more than one point higher NegativeNear@0.20 than local-atlas and improvement on \pptc {}@0.20, \pptc {}@0.40, or mean positive distance. Passing this gate permits rollout evaluation; it is not \outcomeptr {} or measured success}{table.7}{}}
|
| 65 |
\@writefile{lot}{\contentsline {table}{\numberline {8}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{9}{table.8}\protected@file@percent }
|
| 66 |
\newlabel{tab:ctt-val-rollout}{{8}{9}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.8}{}}
|
|
|
|
| 67 |
\@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{10}{table.9}\protected@file@percent }
|
| 68 |
\newlabel{tab:ctt-base-context-obs-val-rollout}{{9}{10}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.9}{}}
|
| 69 |
+
\@writefile{toc}{\contentsline {section}{\numberline {8}Reproducibility Artifacts}{10}{section.8}\protected@file@percent }
|
| 70 |
\@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{11}{table.10}\protected@file@percent }
|
| 71 |
\newlabel{tab:ctt-test-rollout}{{10}{11}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.10}{}}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 72 |
\@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{12}{table.11}\protected@file@percent }
|
| 73 |
\newlabel{tab:ctt-base-context-obs-test-rollout}{{11}{12}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.11}{}}
|
| 74 |
\@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{12}{table.12}\protected@file@percent }
|
| 75 |
\newlabel{tab:ctt-dominance}{{12}{12}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.12}{}}
|
| 76 |
+
\@writefile{toc}{\contentsline {section}{\numberline {9}Limitations and Next Steps}{12}{section.9}\protected@file@percent }
|
| 77 |
+
\bibstyle{plain}
|
| 78 |
+
\bibdata{references}
|
| 79 |
+
\bibcite{chen2025robotwin2}{1}
|
| 80 |
+
\bibcite{glossop2025cast}{2}
|
| 81 |
\bibcite{kim2025openvlaoft}{3}
|
| 82 |
\bibcite{kim2024openvla}{4}
|
| 83 |
\bibcite{kwok2025robomonkey}{5}
|
| 84 |
\bibcite{liu2023libero}{6}
|
| 85 |
\bibcite{singh2026bokbo}{7}
|
| 86 |
\bibcite{tao2024maniskill3}{8}
|
|
|
|
|
|
|
| 87 |
\@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate.}}{13}{table.13}\protected@file@percent }
|
| 88 |
\newlabel{tab:ctt-learned-dominance}{{13}{13}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate}{table.13}{}}
|
| 89 |
\@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{13}{table.14}\protected@file@percent }
|
| 90 |
\newlabel{tab:ctt-base-context-obs-learned-dominance}{{14}{13}{Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.14}{}}
|
| 91 |
+
\@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{13}{section.10}\protected@file@percent }
|
| 92 |
+
\bibcite{zhang2024vlabench}{9}
|
| 93 |
+
\bibcite{zhao2026verispace}{10}
|
| 94 |
+
\@writefile{lot}{\contentsline {table}{\numberline {15}{\ignorespaces Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate.}}{14}{table.15}\protected@file@percent }
|
| 95 |
+
\newlabel{tab:ctt-base-context-obs-learned-train-dominance}{{15}{14}{Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate}{table.15}{}}
|
| 96 |
+
\@writefile{lot}{\contentsline {table}{\numberline {16}{\ignorespaces Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table\nobreakspace {}\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector.}}{14}{table.16}\protected@file@percent }
|
| 97 |
+
\newlabel{tab:ctt-base-context-obs-nonlinear-train-dominance}{{16}{14}{Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table~\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector}{table.16}{}}
|
| 98 |
+
\gdef \@abspage@last{14}
|