| \begin{thebibliography}{} |
| |
| \bibitem[Ambrosio et~al., 2005]{ambrosio2008gradient} |
| Ambrosio, L., Gigli, N., and Savare, G. (2005). |
| \newblock {\em Gradient Flows: {{In}} Metric Spaces and in the Space of Probability Measures}. |
| \newblock Springer Science \& Business Media. |
|
|
| \bibitem[Ba et~al., 2021]{ba2021understanding} |
| Ba, J., Erdogdu, M.~A., Ghassemi, M., Sun, S., Suzuki, T., Wu, D., and Zhang, T. (2021). |
| \newblock Understanding the variance collapse of svgd in high dimensions. |
| \newblock In {\em International Conference on Learning Representations}. |
|
|
| \bibitem[{Ben-Tal} et~al., 2013]{ben-talRobustSolutionsOptimization2013} |
| {Ben-Tal}, A., {den Hertog}, D., De~Waegenaere, A., Melenberg, B., and Rennen, G. (2013). |
| \newblock Robust {{Solutions}} of {{Optimization Problems Affected}} by {{Uncertain Probabilities}}. |
| \newblock {\em Management Science}, 59(2):341--357. |
|
|
| \bibitem[Blanchet and Glynn, 2015]{blanchet2015unbiased} |
| Blanchet, J.~H. and Glynn, P.~W. (2015). |
| \newblock Unbiased monte carlo for optimization and functions of expectations via multi-level randomization. |
| \newblock In {\em 2015 Winter Simulation Conference (WSC)}, pages 3656--3667. IEEE. |
|
|
| \bibitem[Carrillo et~al., 2024]{carrillo2024fisher} |
| Carrillo, J.~A., Chen, Y., Huang, D.~Z., Huang, J., and Wei, D. (2024). |
| \newblock Fisher-rao gradient flow: geodesic convexity and functional inequalities. |
| \newblock {\em arXiv preprint arXiv:2407.15693}. |
|
|
| \bibitem[Chen et~al., 2022]{chen2022improved} |
| Chen, Y., Chewi, S., Salim, A., and Wibisono, A. (2022). |
| \newblock Improved analysis for a proximal algorithm for sampling. |
| \newblock In {\em Conference on Learning Theory}, pages 2984--3014. PMLR. |
|
|
| \bibitem[Chen et~al., 2023]{chen2023sampling} |
| Chen, Y., Huang, D.~Z., Huang, J., Reich, S., and Stuart, A.~M. (2023). |
| \newblock Sampling via gradient flows in the space of probability measures. |
| \newblock {\em arXiv preprint arXiv:2310.03597}. |
|
|
| \bibitem[Chewi et~al., 2022]{chewi2022query} |
| Chewi, S., Gerber, P.~R., Lu, C., Le~Gouic, T., and Rigollet, P. (2022). |
| \newblock The query complexity of sampling from strongly log-concave distributions in one dimension. |
| \newblock In {\em Conference on Learning Theory}, pages 2041--2059. PMLR. |
|
|
| \bibitem[Chewi et~al., 2024]{chewi2024statistical} |
| Chewi, S., Niles-Weed, J., and Rigollet, P. (2024). |
| \newblock Statistical optimal transport. |
| \newblock {\em arXiv preprint arXiv:2407.18163}. |
|
|
| \bibitem[Conger et~al., 2023]{congerStrategicDistributionShift2023} |
| Conger, L.~E., Hoffman, F., Mazumdar, E., and Ratliff, L.~J. (2023). |
| \newblock Strategic {{Distribution Shift}} of {{Interacting Agents}} via {{Coupled Gradient Flows}}. |
| \newblock In {\em Thirty-Seventh {{Conference}} on {{Neural Information Processing Systems}}}. |
|
|
| \bibitem[Delage and Ye, 2010]{delageDistributionallyRobustOptimization2010} |
| Delage, E. and Ye, Y. (2010). |
| \newblock Distributionally robust optimization under moment uncertainty with application to data-driven problems. |
| \newblock {\em Operations research}, 58(3):595--612. |
|
|
| \bibitem[El~Ghaoui and Lebret, 1997]{el1997robust} |
| El~Ghaoui, L. and Lebret, H. (1997). |
| \newblock Robust solutions to least-squares problems with uncertain data. |
| \newblock {\em SIAM Journal on matrix analysis and applications}, 18(4):1035--1064. |
|
|
| \bibitem[Gao and Kleywegt, 2016]{gaoDistributionallyRobustStochastic2016} |
| Gao, R. and Kleywegt, A.~J. (2016). |
| \newblock Distributionally {{Robust Stochastic Optimization}} with {{Wasserstein Distance}}. |
| \newblock {\em arXiv preprint arXiv:1604.02199}. |
|
|
| \bibitem[Garc{\'\i}a~Trillos and Garc{\'\i}a~Trillos, 2024]{trillosAdversarialRobustnessUse2023} |
| Garc{\'\i}a~Trillos, C.~A. and Garc{\'\i}a~Trillos, N. (2024). |
| \newblock On adversarial robustness and the use of wasserstein ascent-descent dynamics to enforce it. |
| \newblock {\em Information and Inference: A Journal of the IMA}, 13(3):iaae018. |
| |
| \bibitem[Garc{\'\i}a~Trillos and Sanz-Alonso, 2018]{garcia2018continuum} |
| Garc{\'\i}a~Trillos, N. and Sanz-Alonso, D. (2018). |
| \newblock Continuum limits of posteriors in graph bayesian inverse problems. |
| \newblock {\em SIAM Journal on Mathematical Analysis}, 50(4):4020--4040. |
| |
| \bibitem[Ghadimi and Lan, 2013]{ghadimi2013stochastic} |
| Ghadimi, S. and Lan, G. (2013). |
| \newblock Stochastic first-and zeroth-order methods for nonconvex stochastic programming. |
| \newblock {\em SIAM journal on optimization}, 23(4):2341--2368. |
| |
| \bibitem[Gross, 1975]{gross1975logarithmic} |
| Gross, L. (1975). |
| \newblock Logarithmic sobolev inequalities. |
| \newblock {\em American Journal of Mathematics}, 97(4):1061--1083. |
| |
| \bibitem[Hu and Hong, 2013]{hu2013kullback} |
| Hu, Z. and Hong, L.~J. (2013). |
| \newblock Kullback-leibler divergence constrained distributionally robust optimization. |
| \newblock {\em Available at Optimization Online}, 1(2):9. |
| |
| \bibitem[Krizhevsky et~al., 2009]{krizhevsky2009learning} |
| Krizhevsky, A., Hinton, G., et~al. (2009). |
| \newblock Learning multiple layers of features from tiny images.(2009). |
| |
| \bibitem[Kuhn et~al., 2025]{kuhnDistributionallyRobustOptimization2024} |
| Kuhn, D., Shafiee, S., and Wiesemann, W. (2025). |
| \newblock Distributionally robust optimization. |
| \newblock {\em Acta Numerica}, 34:579--804. |
| |
| \bibitem[Lee et~al., 2021]{lee2021structured} |
| Lee, Y.~T., Shen, R., and Tian, K. (2021). |
| \newblock Structured logconcave sampling with a restricted gaussian oracle. |
| \newblock In {\em Conference on Learning Theory}, pages 2993--3050. PMLR. |
| |
| \bibitem[Levy et~al., 2020]{levyLargeScaleMethodsDistributionally2020} |
| Levy, D., Carmon, Y., Duchi, J.~C., and Sidford, A. (2020). |
| \newblock Large-scale methods for distributionally robust optimization. |
| \newblock {\em Advances in neural information processing systems}, 33:8847--8860. |
| |
| \bibitem[Liu and Wang, 2016]{liu2016stein} |
| Liu, Q. and Wang, D. (2016). |
| \newblock Stein variational gradient descent: A general purpose bayesian inference algorithm. |
| \newblock {\em Advances in neural information processing systems}, 29. |
| |
| \bibitem[Lu et~al., 2019]{luAcceleratingLangevinSampling2019} |
| Lu, Y., Lu, J., and Nolen, J. (2019). |
| \newblock Accelerating langevin sampling with birth-death. |
| \newblock {\em arXiv preprint arXiv:1905.09863}. |
| |
| \bibitem[Lu et~al., 2023]{luBirthdeathDynamicsSampling2023} |
| Lu, Y., Slep{\v c}ev, D., and Wang, L. (2023). |
| \newblock Birth-death dynamics for sampling: {{Global}} convergence, approximations and their asymptotics. |
| \newblock {\em Nonlinearity}, 36(11):5731--5772. |
| |
| \bibitem[Mielke, 2023]{mielke2023introduction} |
| Mielke, A. (2023). |
| \newblock An introduction to the analysis of gradients systems. |
| \newblock {\em arXiv preprint arXiv:2306.05026}. |
| |
| \bibitem[Mielke, 2025]{mielkeNotesHellingerDistance2025} |
| Mielke, A. (2025). |
| \newblock Some notes on the hellinger distance and various fisher-rao distances. |
| \newblock {\em arXiv preprint arXiv:2510.02537}. |
| |
| \bibitem[Mielke and Zhu, 2025]{mielke2025hellinger} |
| Mielke, A. and Zhu, J.-J. (2025). |
| \newblock Hellinger-kantorovich gradient flows: Global exponential decay of entropy functionals. |
| \newblock {\em arXiv preprint arXiv:2501.17049}. |
| |
| \bibitem[Mohajerin~Esfahani and Kuhn, 2018]{mohajerin2018data} |
| Mohajerin~Esfahani, P. and Kuhn, D. (2018). |
| \newblock Data-driven distributionally robust optimization using the wasserstein metric: Performance guarantees and tractable reformulations. |
| \newblock {\em Mathematical Programming}, 171(1):115--166. |
| |
| \bibitem[Otto, 1996]{otto1996double} |
| Otto, F. (1996). |
| \newblock {\em Double degenerate diffusion equations as steepest descent}. |
| \newblock Sonderforschungsbereich 256. |
| |
| \bibitem[Salim et~al., 2022]{salim2022convergence} |
| Salim, A., Sun, L., and Richtarik, P. (2022). |
| \newblock A convergence theory for svgd in the population limit under talagrand’s inequality t1. |
| \newblock In {\em International Conference on Machine Learning}, pages 19139--19152. PMLR. |
| |
| \bibitem[Shi and Mackey, 2023]{shi2023finite} |
| Shi, J. and Mackey, L. (2023). |
| \newblock A finite-particle convergence rate for stein variational gradient descent. |
| \newblock {\em Advances in Neural Information Processing Systems}, 36:26831--26844. |
| |
| \bibitem[Sinha et~al., 2017]{sinha2020certifyingdistributionalrobustnessprincipled} |
| Sinha, A., Namkoong, H., Volpi, R., and Duchi, J. (2017). |
| \newblock Certifying some distributional robustness with principled adversarial training. |
| \newblock {\em arXiv preprint arXiv:1710.10571}. |
| |
| \bibitem[Vempala and Wibisono, 2019]{vempala2019rapid} |
| Vempala, S. and Wibisono, A. (2019). |
| \newblock Rapid convergence of the unadjusted langevin algorithm: Isoperimetry suffices. |
| \newblock {\em Advances in neural information processing systems}, 32. |
| |
| \bibitem[Wang and Chizat, 2022]{wang2022exponentially} |
| Wang, G. and Chizat, L. (2022). |
| \newblock An exponentially converging particle method for the mixed nash equilibrium of continuous games. |
| \newblock {\em arXiv preprint arXiv:2211.01280}. |
| |
| \bibitem[Wang et~al., 2021]{wang2021sinkhorn} |
| Wang, J., Gao, R., and Xie, Y. (2021). |
| \newblock Sinkhorn distributionally robust optimization. |
| \newblock {\em arXiv preprint arXiv:2109.11926}. |
| |
| \bibitem[Wibisono, 2025]{wibisono2025mixing} |
| Wibisono, A. (2025). |
| \newblock Mixing time of the proximal sampler in relative fisher information via strong data processing inequality. |
| \newblock {\em arXiv preprint arXiv:2502.05623}. |
| |
| \bibitem[Xu et~al., 2024]{xu2024flow} |
| Xu, C., Lee, J., Cheng, X., and Xie, Y. (2024). |
| \newblock Flow-based distributionally robust optimization. |
| \newblock {\em IEEE Journal on Selected Areas in Information Theory}, 5:62--77. |
| |
| \bibitem[Yu et~al., 2022]{yuFastDistributionallyRobust2022} |
| Yu, Y., Lin, T., Mazumdar, E.~V., and Jordan, M. (2022). |
| \newblock Fast {{Distributionally Robust Learning}} with {{Variance-Reduced Min-Max Optimization}}. |
| \newblock In {\em Proceedings of {{The}} 25th {{International Conference}} on {{Artificial Intelligence}} and {{Statistics}}}, pages 1219--1250. PMLR. |
| |
| \bibitem[Zhao and Guan, 2018]{zhaoDatadrivenRiskaverseStochastic2018} |
| Zhao, C. and Guan, Y. (2018). |
| \newblock Data-driven risk-averse stochastic optimization with {{Wasserstein}} metric. |
| \newblock {\em Operations Research Letters}, 46(2):262--267. |
| |
| \bibitem[Zhu et~al., 2021]{zhu2021kernel} |
| Zhu, J.-J., Jitkrittum, W., Diehl, M., and Sch{\"o}lkopf, B. (2021). |
| \newblock Kernel distributionally robust optimization: Generalized duality theorem and stochastic approximation. |
| \newblock In {\em International Conference on Artificial Intelligence and Statistics}, pages 280--288. PMLR. |
| |
| \bibitem[Zhu and Xie, 2024]{zhu2024distributionally} |
| Zhu, L. and Xie, Y. (2024). |
| \newblock Distributionally robust optimization via iterative algorithms in continuous probability spaces. |
| \newblock {\em arXiv preprint arXiv:2412.20556}. |
| |
| \end{thebibliography} |
| |