File size: 10,661 Bytes
cbeeed4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
\begin{thebibliography}{}

\bibitem[Ambrosio et~al., 2005]{ambrosio2008gradient}
Ambrosio, L., Gigli, N., and Savare, G. (2005).
\newblock {\em Gradient Flows: {{In}} Metric Spaces and in the Space of Probability Measures}.
\newblock Springer Science \& Business Media.

\bibitem[Ba et~al., 2021]{ba2021understanding}
Ba, J., Erdogdu, M.~A., Ghassemi, M., Sun, S., Suzuki, T., Wu, D., and Zhang, T. (2021).
\newblock Understanding the variance collapse of svgd in high dimensions.
\newblock In {\em International Conference on Learning Representations}.

\bibitem[{Ben-Tal} et~al., 2013]{ben-talRobustSolutionsOptimization2013}
{Ben-Tal}, A., {den Hertog}, D., De~Waegenaere, A., Melenberg, B., and Rennen, G. (2013).
\newblock Robust {{Solutions}} of {{Optimization Problems Affected}} by {{Uncertain Probabilities}}.
\newblock {\em Management Science}, 59(2):341--357.

\bibitem[Blanchet and Glynn, 2015]{blanchet2015unbiased}
Blanchet, J.~H. and Glynn, P.~W. (2015).
\newblock Unbiased monte carlo for optimization and functions of expectations via multi-level randomization.
\newblock In {\em 2015 Winter Simulation Conference (WSC)}, pages 3656--3667. IEEE.

\bibitem[Carrillo et~al., 2024]{carrillo2024fisher}
Carrillo, J.~A., Chen, Y., Huang, D.~Z., Huang, J., and Wei, D. (2024).
\newblock Fisher-rao gradient flow: geodesic convexity and functional inequalities.
\newblock {\em arXiv preprint arXiv:2407.15693}.

\bibitem[Chen et~al., 2022]{chen2022improved}
Chen, Y., Chewi, S., Salim, A., and Wibisono, A. (2022).
\newblock Improved analysis for a proximal algorithm for sampling.
\newblock In {\em Conference on Learning Theory}, pages 2984--3014. PMLR.

\bibitem[Chen et~al., 2023]{chen2023sampling}
Chen, Y., Huang, D.~Z., Huang, J., Reich, S., and Stuart, A.~M. (2023).
\newblock Sampling via gradient flows in the space of probability measures.
\newblock {\em arXiv preprint arXiv:2310.03597}.

\bibitem[Chewi et~al., 2022]{chewi2022query}
Chewi, S., Gerber, P.~R., Lu, C., Le~Gouic, T., and Rigollet, P. (2022).
\newblock The query complexity of sampling from strongly log-concave distributions in one dimension.
\newblock In {\em Conference on Learning Theory}, pages 2041--2059. PMLR.

\bibitem[Chewi et~al., 2024]{chewi2024statistical}
Chewi, S., Niles-Weed, J., and Rigollet, P. (2024).
\newblock Statistical optimal transport.
\newblock {\em arXiv preprint arXiv:2407.18163}.

\bibitem[Conger et~al., 2023]{congerStrategicDistributionShift2023}
Conger, L.~E., Hoffman, F., Mazumdar, E., and Ratliff, L.~J. (2023).
\newblock Strategic {{Distribution Shift}} of {{Interacting Agents}} via {{Coupled Gradient Flows}}.
\newblock In {\em Thirty-Seventh {{Conference}} on {{Neural Information Processing Systems}}}.

\bibitem[Delage and Ye, 2010]{delageDistributionallyRobustOptimization2010}
Delage, E. and Ye, Y. (2010).
\newblock Distributionally robust optimization under moment uncertainty with application to data-driven problems.
\newblock {\em Operations research}, 58(3):595--612.

\bibitem[El~Ghaoui and Lebret, 1997]{el1997robust}
El~Ghaoui, L. and Lebret, H. (1997).
\newblock Robust solutions to least-squares problems with uncertain data.
\newblock {\em SIAM Journal on matrix analysis and applications}, 18(4):1035--1064.

\bibitem[Gao and Kleywegt, 2016]{gaoDistributionallyRobustStochastic2016}
Gao, R. and Kleywegt, A.~J. (2016).
\newblock Distributionally {{Robust Stochastic Optimization}} with {{Wasserstein Distance}}.
\newblock {\em arXiv preprint arXiv:1604.02199}.

\bibitem[Garc{\'\i}a~Trillos and Garc{\'\i}a~Trillos, 2024]{trillosAdversarialRobustnessUse2023}
Garc{\'\i}a~Trillos, C.~A. and Garc{\'\i}a~Trillos, N. (2024).
\newblock On adversarial robustness and the use of wasserstein ascent-descent dynamics to enforce it.
\newblock {\em Information and Inference: A Journal of the IMA}, 13(3):iaae018.

\bibitem[Garc{\'\i}a~Trillos and Sanz-Alonso, 2018]{garcia2018continuum}
Garc{\'\i}a~Trillos, N. and Sanz-Alonso, D. (2018).
\newblock Continuum limits of posteriors in graph bayesian inverse problems.
\newblock {\em SIAM Journal on Mathematical Analysis}, 50(4):4020--4040.

\bibitem[Ghadimi and Lan, 2013]{ghadimi2013stochastic}
Ghadimi, S. and Lan, G. (2013).
\newblock Stochastic first-and zeroth-order methods for nonconvex stochastic programming.
\newblock {\em SIAM journal on optimization}, 23(4):2341--2368.

\bibitem[Gross, 1975]{gross1975logarithmic}
Gross, L. (1975).
\newblock Logarithmic sobolev inequalities.
\newblock {\em American Journal of Mathematics}, 97(4):1061--1083.

\bibitem[Hu and Hong, 2013]{hu2013kullback}
Hu, Z. and Hong, L.~J. (2013).
\newblock Kullback-leibler divergence constrained distributionally robust optimization.
\newblock {\em Available at Optimization Online}, 1(2):9.

\bibitem[Krizhevsky et~al., 2009]{krizhevsky2009learning}
Krizhevsky, A., Hinton, G., et~al. (2009).
\newblock Learning multiple layers of features from tiny images.(2009).

\bibitem[Kuhn et~al., 2025]{kuhnDistributionallyRobustOptimization2024}
Kuhn, D., Shafiee, S., and Wiesemann, W. (2025).
\newblock Distributionally robust optimization.
\newblock {\em Acta Numerica}, 34:579--804.

\bibitem[Lee et~al., 2021]{lee2021structured}
Lee, Y.~T., Shen, R., and Tian, K. (2021).
\newblock Structured logconcave sampling with a restricted gaussian oracle.
\newblock In {\em Conference on Learning Theory}, pages 2993--3050. PMLR.

\bibitem[Levy et~al., 2020]{levyLargeScaleMethodsDistributionally2020}
Levy, D., Carmon, Y., Duchi, J.~C., and Sidford, A. (2020).
\newblock Large-scale methods for distributionally robust optimization.
\newblock {\em Advances in neural information processing systems}, 33:8847--8860.

\bibitem[Liu and Wang, 2016]{liu2016stein}
Liu, Q. and Wang, D. (2016).
\newblock Stein variational gradient descent: A general purpose bayesian inference algorithm.
\newblock {\em Advances in neural information processing systems}, 29.

\bibitem[Lu et~al., 2019]{luAcceleratingLangevinSampling2019}
Lu, Y., Lu, J., and Nolen, J. (2019).
\newblock Accelerating langevin sampling with birth-death.
\newblock {\em arXiv preprint arXiv:1905.09863}.

\bibitem[Lu et~al., 2023]{luBirthdeathDynamicsSampling2023}
Lu, Y., Slep{\v c}ev, D., and Wang, L. (2023).
\newblock Birth-death dynamics for sampling: {{Global}} convergence, approximations and their asymptotics.
\newblock {\em Nonlinearity}, 36(11):5731--5772.

\bibitem[Mielke, 2023]{mielke2023introduction}
Mielke, A. (2023).
\newblock An introduction to the analysis of gradients systems.
\newblock {\em arXiv preprint arXiv:2306.05026}.

\bibitem[Mielke, 2025]{mielkeNotesHellingerDistance2025}
Mielke, A. (2025).
\newblock Some notes on the hellinger distance and various fisher-rao distances.
\newblock {\em arXiv preprint arXiv:2510.02537}.

\bibitem[Mielke and Zhu, 2025]{mielke2025hellinger}
Mielke, A. and Zhu, J.-J. (2025).
\newblock Hellinger-kantorovich gradient flows: Global exponential decay of entropy functionals.
\newblock {\em arXiv preprint arXiv:2501.17049}.

\bibitem[Mohajerin~Esfahani and Kuhn, 2018]{mohajerin2018data}
Mohajerin~Esfahani, P. and Kuhn, D. (2018).
\newblock Data-driven distributionally robust optimization using the wasserstein metric: Performance guarantees and tractable reformulations.
\newblock {\em Mathematical Programming}, 171(1):115--166.

\bibitem[Otto, 1996]{otto1996double}
Otto, F. (1996).
\newblock {\em Double degenerate diffusion equations as steepest descent}.
\newblock Sonderforschungsbereich 256.

\bibitem[Salim et~al., 2022]{salim2022convergence}
Salim, A., Sun, L., and Richtarik, P. (2022).
\newblock A convergence theory for svgd in the population limit under talagrand’s inequality t1.
\newblock In {\em International Conference on Machine Learning}, pages 19139--19152. PMLR.

\bibitem[Shi and Mackey, 2023]{shi2023finite}
Shi, J. and Mackey, L. (2023).
\newblock A finite-particle convergence rate for stein variational gradient descent.
\newblock {\em Advances in Neural Information Processing Systems}, 36:26831--26844.

\bibitem[Sinha et~al., 2017]{sinha2020certifyingdistributionalrobustnessprincipled}
Sinha, A., Namkoong, H., Volpi, R., and Duchi, J. (2017).
\newblock Certifying some distributional robustness with principled adversarial training.
\newblock {\em arXiv preprint arXiv:1710.10571}.

\bibitem[Vempala and Wibisono, 2019]{vempala2019rapid}
Vempala, S. and Wibisono, A. (2019).
\newblock Rapid convergence of the unadjusted langevin algorithm: Isoperimetry suffices.
\newblock {\em Advances in neural information processing systems}, 32.

\bibitem[Wang and Chizat, 2022]{wang2022exponentially}
Wang, G. and Chizat, L. (2022).
\newblock An exponentially converging particle method for the mixed nash equilibrium of continuous games.
\newblock {\em arXiv preprint arXiv:2211.01280}.

\bibitem[Wang et~al., 2021]{wang2021sinkhorn}
Wang, J., Gao, R., and Xie, Y. (2021).
\newblock Sinkhorn distributionally robust optimization.
\newblock {\em arXiv preprint arXiv:2109.11926}.

\bibitem[Wibisono, 2025]{wibisono2025mixing}
Wibisono, A. (2025).
\newblock Mixing time of the proximal sampler in relative fisher information via strong data processing inequality.
\newblock {\em arXiv preprint arXiv:2502.05623}.

\bibitem[Xu et~al., 2024]{xu2024flow}
Xu, C., Lee, J., Cheng, X., and Xie, Y. (2024).
\newblock Flow-based distributionally robust optimization.
\newblock {\em IEEE Journal on Selected Areas in Information Theory}, 5:62--77.

\bibitem[Yu et~al., 2022]{yuFastDistributionallyRobust2022}
Yu, Y., Lin, T., Mazumdar, E.~V., and Jordan, M. (2022).
\newblock Fast {{Distributionally Robust Learning}} with {{Variance-Reduced Min-Max Optimization}}.
\newblock In {\em Proceedings of {{The}} 25th {{International Conference}} on {{Artificial Intelligence}} and {{Statistics}}}, pages 1219--1250. PMLR.

\bibitem[Zhao and Guan, 2018]{zhaoDatadrivenRiskaverseStochastic2018}
Zhao, C. and Guan, Y. (2018).
\newblock Data-driven risk-averse stochastic optimization with {{Wasserstein}} metric.
\newblock {\em Operations Research Letters}, 46(2):262--267.

\bibitem[Zhu et~al., 2021]{zhu2021kernel}
Zhu, J.-J., Jitkrittum, W., Diehl, M., and Sch{\"o}lkopf, B. (2021).
\newblock Kernel distributionally robust optimization: Generalized duality theorem and stochastic approximation.
\newblock In {\em International Conference on Artificial Intelligence and Statistics}, pages 280--288. PMLR.

\bibitem[Zhu and Xie, 2024]{zhu2024distributionally}
Zhu, L. and Xie, Y. (2024).
\newblock Distributionally robust optimization via iterative algorithms in continuous probability spaces.
\newblock {\em arXiv preprint arXiv:2412.20556}.

\end{thebibliography}