anhtld commited on
Commit
84b30d7
·
verified ·
1 Parent(s): 18a6ef2

auto-sync 2026-07-03T16:53:36Z workspace

Browse files
Files changed (1) hide show
  1. workspace/latex/main.aux +14 -10
workspace/latex/main.aux CHANGED
@@ -64,31 +64,35 @@
64
  \newlabel{tab:ctt-val-proxy}{{7}{8}{Validation proxy comparison on 69 validation charts with measured positive tangents, using train-only source positives. The gate column is proxy-only: it requires no more than one point higher NegativeNear@0.20 than local-atlas and improvement on \pptc {}@0.20, \pptc {}@0.40, or mean positive distance. Passing this gate permits rollout evaluation; it is not \outcomeptr {} or measured success}{table.7}{}}
65
  \@writefile{lot}{\contentsline {table}{\numberline {8}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{9}{table.8}\protected@file@percent }
66
  \newlabel{tab:ctt-val-rollout}{{8}{9}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.8}{}}
67
- \@writefile{toc}{\contentsline {section}{\numberline {8}Reproducibility Artifacts}{9}{section.8}\protected@file@percent }
68
  \@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{10}{table.9}\protected@file@percent }
69
  \newlabel{tab:ctt-base-context-obs-val-rollout}{{9}{10}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.9}{}}
 
70
  \@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{11}{table.10}\protected@file@percent }
71
  \newlabel{tab:ctt-test-rollout}{{10}{11}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.10}{}}
72
- \@writefile{toc}{\contentsline {section}{\numberline {9}Limitations and Next Steps}{11}{section.9}\protected@file@percent }
73
- \bibstyle{plain}
74
- \bibdata{references}
75
- \bibcite{chen2025robotwin2}{1}
76
- \bibcite{glossop2025cast}{2}
77
  \@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{12}{table.11}\protected@file@percent }
78
  \newlabel{tab:ctt-base-context-obs-test-rollout}{{11}{12}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.11}{}}
79
  \@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{12}{table.12}\protected@file@percent }
80
  \newlabel{tab:ctt-dominance}{{12}{12}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.12}{}}
81
- \@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{12}{section.10}\protected@file@percent }
 
 
 
 
82
  \bibcite{kim2025openvlaoft}{3}
83
  \bibcite{kim2024openvla}{4}
84
  \bibcite{kwok2025robomonkey}{5}
85
  \bibcite{liu2023libero}{6}
86
  \bibcite{singh2026bokbo}{7}
87
  \bibcite{tao2024maniskill3}{8}
88
- \bibcite{zhang2024vlabench}{9}
89
- \bibcite{zhao2026verispace}{10}
90
  \@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate.}}{13}{table.13}\protected@file@percent }
91
  \newlabel{tab:ctt-learned-dominance}{{13}{13}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate}{table.13}{}}
92
  \@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{13}{table.14}\protected@file@percent }
93
  \newlabel{tab:ctt-base-context-obs-learned-dominance}{{14}{13}{Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.14}{}}
94
- \gdef \@abspage@last{13}
 
 
 
 
 
 
 
 
64
  \newlabel{tab:ctt-val-proxy}{{7}{8}{Validation proxy comparison on 69 validation charts with measured positive tangents, using train-only source positives. The gate column is proxy-only: it requires no more than one point higher NegativeNear@0.20 than local-atlas and improvement on \pptc {}@0.20, \pptc {}@0.40, or mean positive distance. Passing this gate permits rollout evaluation; it is not \outcomeptr {} or measured success}{table.7}{}}
65
  \@writefile{lot}{\contentsline {table}{\numberline {8}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \ensuremath {\mathrm {PPTC}}{} proxies.}}{9}{table.8}\protected@file@percent }
66
  \newlabel{tab:ctt-val-rollout}{{8}{9}{Measured residual \ctt {} rollout on 69 validation positive-support charts across three train seeds, K=8. Generated candidates are decoded and executed from restored simulator states; these are measured outcome metrics, not \pptc {} proxies}{table.8}{}}
 
67
  \@writefile{lot}{\contentsline {table}{\numberline {9}{\ignorespaces Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action.}}{10}{table.9}\protected@file@percent }
68
  \newlabel{tab:ctt-base-context-obs-val-rollout}{{9}{10}{Measured validation rollout for the proxy-positive \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. The RGB-stat chart token improves support-side validation metrics, but the selected action still fails to beat the base action}{table.9}{}}
69
+ \@writefile{toc}{\contentsline {section}{\numberline {8}Reproducibility Artifacts}{10}{section.8}\protected@file@percent }
70
  \@writefile{lot}{\contentsline {table}{\numberline {10}{\ignorespaces Measured residual \textsc {CTT}{} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates.}}{11}{table.10}\protected@file@percent }
71
  \newlabel{tab:ctt-test-rollout}{{10}{11}{Measured residual \ctt {} rollout on 48 test positive-support charts across three train seeds, K=8. The generated proposal oracle crosses the internal 50\% support target, but the selected action fails because the current score/dominance rule chooses poor candidates}{table.10}{}}
 
 
 
 
 
72
  \@writefile{lot}{\contentsline {table}{\numberline {11}{\ignorespaces Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base.}}{12}{table.11}\protected@file@percent }
73
  \newlabel{tab:ctt-base-context-obs-test-rollout}{{11}{12}{Measured test rollout for the \texttt {base\_context\_obs} visual-stat chart token, across three train seeds, K=8. Support and score-only selection improve relative to the base-action chart token, but selected success remains below base}{table.11}{}}
74
  \@writefile{lot}{\contentsline {table}{\numberline {12}{\ignorespaces Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer.}}{12}{table.12}\protected@file@percent }
75
  \newlabel{tab:ctt-dominance}{{12}{12}{Validation-calibrated dominance fallback evaluated on the measured test rollout. The threshold and conformal residual quantile are fit on validation rows only; test outcomes are used only for reporting. The fallback reduces coverage but does not yet repair selector transfer}{table.12}{}}
76
+ \@writefile{toc}{\contentsline {section}{\numberline {9}Limitations and Next Steps}{12}{section.9}\protected@file@percent }
77
+ \bibstyle{plain}
78
+ \bibdata{references}
79
+ \bibcite{chen2025robotwin2}{1}
80
+ \bibcite{glossop2025cast}{2}
81
  \bibcite{kim2025openvlaoft}{3}
82
  \bibcite{kim2024openvla}{4}
83
  \bibcite{kwok2025robomonkey}{5}
84
  \bibcite{liu2023libero}{6}
85
  \bibcite{singh2026bokbo}{7}
86
  \bibcite{tao2024maniskill3}{8}
 
 
87
  \@writefile{lot}{\contentsline {table}{\numberline {13}{\ignorespaces Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate.}}{13}{table.13}\protected@file@percent }
88
  \newlabel{tab:ctt-learned-dominance}{{13}{13}{Learned dominance fallback trained on validation measured rows and evaluated on held-out test rows. Features are deployment-visible candidate features: utility-energy scores, score margins to base, rank, and tangent norms. This improves over the base test success but remains far below the paper gate}{table.13}{}}
89
  \@writefile{lot}{\contentsline {table}{\numberline {14}{\ignorespaces Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows.}}{13}{table.14}\protected@file@percent }
90
  \newlabel{tab:ctt-base-context-obs-learned-dominance}{{14}{13}{Best validation-calibrated dominance diagnostic so far: learned context dominance over the measured \texttt {base\_context\_obs} visual-stat rollout rows. The calibrator is fit on validation measured rows only and evaluated on held-out test rows}{table.14}{}}
91
+ \@writefile{toc}{\contentsline {section}{\numberline {10}Conclusion}{13}{section.10}\protected@file@percent }
92
+ \bibcite{zhang2024vlabench}{9}
93
+ \bibcite{zhao2026verispace}{10}
94
+ \@writefile{lot}{\contentsline {table}{\numberline {15}{\ignorespaces Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate.}}{14}{table.15}\protected@file@percent }
95
+ \newlabel{tab:ctt-base-context-obs-learned-train-dominance}{{15}{14}{Train-calibrated learned dominance evaluated on the same held-out test rollout rows. Calibration uses only train-split measured generated rollouts with same-chart and same-state source retrieval excluded. This is a cleaner selector diagnostic than validation calibration, but it does not beat the validation-calibrated best row and still fails the deployment gate}{table.15}{}}
96
+ \@writefile{lot}{\contentsline {table}{\numberline {16}{\ignorespaces Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table\nobreakspace {}\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector.}}{14}{table.16}\protected@file@percent }
97
+ \newlabel{tab:ctt-base-context-obs-nonlinear-train-dominance}{{16}{14}{Nonlinear train-calibrated selector diagnostic. The model and threshold are selected only on held-out train-calibration rows, then evaluated once on the held-out test rollout rows. The best nonlinear row does not beat Table~\ref {tab:ctt-base-context-obs-learned-train-dominance}, so the current bottleneck is not just linear separability in the dominance selector}{table.16}{}}
98
+ \gdef \@abspage@last{14}