\begin{thebibliography}{} \bibitem[Ambrosio et~al., 2005]{ambrosio2008gradient} Ambrosio, L., Gigli, N., and Savare, G. (2005). \newblock {\em Gradient Flows: {{In}} Metric Spaces and in the Space of Probability Measures}. \newblock Springer Science \& Business Media. \bibitem[Ba et~al., 2021]{ba2021understanding} Ba, J., Erdogdu, M.~A., Ghassemi, M., Sun, S., Suzuki, T., Wu, D., and Zhang, T. (2021). \newblock Understanding the variance collapse of svgd in high dimensions. \newblock In {\em International Conference on Learning Representations}. \bibitem[{Ben-Tal} et~al., 2013]{ben-talRobustSolutionsOptimization2013} {Ben-Tal}, A., {den Hertog}, D., De~Waegenaere, A., Melenberg, B., and Rennen, G. (2013). \newblock Robust {{Solutions}} of {{Optimization Problems Affected}} by {{Uncertain Probabilities}}. \newblock {\em Management Science}, 59(2):341--357. \bibitem[Blanchet and Glynn, 2015]{blanchet2015unbiased} Blanchet, J.~H. and Glynn, P.~W. (2015). \newblock Unbiased monte carlo for optimization and functions of expectations via multi-level randomization. \newblock In {\em 2015 Winter Simulation Conference (WSC)}, pages 3656--3667. IEEE. \bibitem[Carrillo et~al., 2024]{carrillo2024fisher} Carrillo, J.~A., Chen, Y., Huang, D.~Z., Huang, J., and Wei, D. (2024). \newblock Fisher-rao gradient flow: geodesic convexity and functional inequalities. \newblock {\em arXiv preprint arXiv:2407.15693}. \bibitem[Chen et~al., 2022]{chen2022improved} Chen, Y., Chewi, S., Salim, A., and Wibisono, A. (2022). \newblock Improved analysis for a proximal algorithm for sampling. \newblock In {\em Conference on Learning Theory}, pages 2984--3014. PMLR. \bibitem[Chen et~al., 2023]{chen2023sampling} Chen, Y., Huang, D.~Z., Huang, J., Reich, S., and Stuart, A.~M. (2023). \newblock Sampling via gradient flows in the space of probability measures. \newblock {\em arXiv preprint arXiv:2310.03597}. \bibitem[Chewi et~al., 2022]{chewi2022query} Chewi, S., Gerber, P.~R., Lu, C., Le~Gouic, T., and Rigollet, P. (2022). \newblock The query complexity of sampling from strongly log-concave distributions in one dimension. \newblock In {\em Conference on Learning Theory}, pages 2041--2059. PMLR. \bibitem[Chewi et~al., 2024]{chewi2024statistical} Chewi, S., Niles-Weed, J., and Rigollet, P. (2024). \newblock Statistical optimal transport. \newblock {\em arXiv preprint arXiv:2407.18163}. \bibitem[Conger et~al., 2023]{congerStrategicDistributionShift2023} Conger, L.~E., Hoffman, F., Mazumdar, E., and Ratliff, L.~J. (2023). \newblock Strategic {{Distribution Shift}} of {{Interacting Agents}} via {{Coupled Gradient Flows}}. \newblock In {\em Thirty-Seventh {{Conference}} on {{Neural Information Processing Systems}}}. \bibitem[Delage and Ye, 2010]{delageDistributionallyRobustOptimization2010} Delage, E. and Ye, Y. (2010). \newblock Distributionally robust optimization under moment uncertainty with application to data-driven problems. \newblock {\em Operations research}, 58(3):595--612. \bibitem[El~Ghaoui and Lebret, 1997]{el1997robust} El~Ghaoui, L. and Lebret, H. (1997). \newblock Robust solutions to least-squares problems with uncertain data. \newblock {\em SIAM Journal on matrix analysis and applications}, 18(4):1035--1064. \bibitem[Gao and Kleywegt, 2016]{gaoDistributionallyRobustStochastic2016} Gao, R. and Kleywegt, A.~J. (2016). \newblock Distributionally {{Robust Stochastic Optimization}} with {{Wasserstein Distance}}. \newblock {\em arXiv preprint arXiv:1604.02199}. \bibitem[Garc{\'\i}a~Trillos and Garc{\'\i}a~Trillos, 2024]{trillosAdversarialRobustnessUse2023} Garc{\'\i}a~Trillos, C.~A. and Garc{\'\i}a~Trillos, N. (2024). \newblock On adversarial robustness and the use of wasserstein ascent-descent dynamics to enforce it. \newblock {\em Information and Inference: A Journal of the IMA}, 13(3):iaae018. \bibitem[Garc{\'\i}a~Trillos and Sanz-Alonso, 2018]{garcia2018continuum} Garc{\'\i}a~Trillos, N. and Sanz-Alonso, D. (2018). \newblock Continuum limits of posteriors in graph bayesian inverse problems. \newblock {\em SIAM Journal on Mathematical Analysis}, 50(4):4020--4040. \bibitem[Ghadimi and Lan, 2013]{ghadimi2013stochastic} Ghadimi, S. and Lan, G. (2013). \newblock Stochastic first-and zeroth-order methods for nonconvex stochastic programming. \newblock {\em SIAM journal on optimization}, 23(4):2341--2368. \bibitem[Gross, 1975]{gross1975logarithmic} Gross, L. (1975). \newblock Logarithmic sobolev inequalities. \newblock {\em American Journal of Mathematics}, 97(4):1061--1083. \bibitem[Hu and Hong, 2013]{hu2013kullback} Hu, Z. and Hong, L.~J. (2013). \newblock Kullback-leibler divergence constrained distributionally robust optimization. \newblock {\em Available at Optimization Online}, 1(2):9. \bibitem[Krizhevsky et~al., 2009]{krizhevsky2009learning} Krizhevsky, A., Hinton, G., et~al. (2009). \newblock Learning multiple layers of features from tiny images.(2009). \bibitem[Kuhn et~al., 2025]{kuhnDistributionallyRobustOptimization2024} Kuhn, D., Shafiee, S., and Wiesemann, W. (2025). \newblock Distributionally robust optimization. \newblock {\em Acta Numerica}, 34:579--804. \bibitem[Lee et~al., 2021]{lee2021structured} Lee, Y.~T., Shen, R., and Tian, K. (2021). \newblock Structured logconcave sampling with a restricted gaussian oracle. \newblock In {\em Conference on Learning Theory}, pages 2993--3050. PMLR. \bibitem[Levy et~al., 2020]{levyLargeScaleMethodsDistributionally2020} Levy, D., Carmon, Y., Duchi, J.~C., and Sidford, A. (2020). \newblock Large-scale methods for distributionally robust optimization. \newblock {\em Advances in neural information processing systems}, 33:8847--8860. \bibitem[Liu and Wang, 2016]{liu2016stein} Liu, Q. and Wang, D. (2016). \newblock Stein variational gradient descent: A general purpose bayesian inference algorithm. \newblock {\em Advances in neural information processing systems}, 29. \bibitem[Lu et~al., 2019]{luAcceleratingLangevinSampling2019} Lu, Y., Lu, J., and Nolen, J. (2019). \newblock Accelerating langevin sampling with birth-death. \newblock {\em arXiv preprint arXiv:1905.09863}. \bibitem[Lu et~al., 2023]{luBirthdeathDynamicsSampling2023} Lu, Y., Slep{\v c}ev, D., and Wang, L. (2023). \newblock Birth-death dynamics for sampling: {{Global}} convergence, approximations and their asymptotics. \newblock {\em Nonlinearity}, 36(11):5731--5772. \bibitem[Mielke, 2023]{mielke2023introduction} Mielke, A. (2023). \newblock An introduction to the analysis of gradients systems. \newblock {\em arXiv preprint arXiv:2306.05026}. \bibitem[Mielke, 2025]{mielkeNotesHellingerDistance2025} Mielke, A. (2025). \newblock Some notes on the hellinger distance and various fisher-rao distances. \newblock {\em arXiv preprint arXiv:2510.02537}. \bibitem[Mielke and Zhu, 2025]{mielke2025hellinger} Mielke, A. and Zhu, J.-J. (2025). \newblock Hellinger-kantorovich gradient flows: Global exponential decay of entropy functionals. \newblock {\em arXiv preprint arXiv:2501.17049}. \bibitem[Mohajerin~Esfahani and Kuhn, 2018]{mohajerin2018data} Mohajerin~Esfahani, P. and Kuhn, D. (2018). \newblock Data-driven distributionally robust optimization using the wasserstein metric: Performance guarantees and tractable reformulations. \newblock {\em Mathematical Programming}, 171(1):115--166. \bibitem[Otto, 1996]{otto1996double} Otto, F. (1996). \newblock {\em Double degenerate diffusion equations as steepest descent}. \newblock Sonderforschungsbereich 256. \bibitem[Salim et~al., 2022]{salim2022convergence} Salim, A., Sun, L., and Richtarik, P. (2022). \newblock A convergence theory for svgd in the population limit under talagrand’s inequality t1. \newblock In {\em International Conference on Machine Learning}, pages 19139--19152. PMLR. \bibitem[Shi and Mackey, 2023]{shi2023finite} Shi, J. and Mackey, L. (2023). \newblock A finite-particle convergence rate for stein variational gradient descent. \newblock {\em Advances in Neural Information Processing Systems}, 36:26831--26844. \bibitem[Sinha et~al., 2017]{sinha2020certifyingdistributionalrobustnessprincipled} Sinha, A., Namkoong, H., Volpi, R., and Duchi, J. (2017). \newblock Certifying some distributional robustness with principled adversarial training. \newblock {\em arXiv preprint arXiv:1710.10571}. \bibitem[Vempala and Wibisono, 2019]{vempala2019rapid} Vempala, S. and Wibisono, A. (2019). \newblock Rapid convergence of the unadjusted langevin algorithm: Isoperimetry suffices. \newblock {\em Advances in neural information processing systems}, 32. \bibitem[Wang and Chizat, 2022]{wang2022exponentially} Wang, G. and Chizat, L. (2022). \newblock An exponentially converging particle method for the mixed nash equilibrium of continuous games. \newblock {\em arXiv preprint arXiv:2211.01280}. \bibitem[Wang et~al., 2021]{wang2021sinkhorn} Wang, J., Gao, R., and Xie, Y. (2021). \newblock Sinkhorn distributionally robust optimization. \newblock {\em arXiv preprint arXiv:2109.11926}. \bibitem[Wibisono, 2025]{wibisono2025mixing} Wibisono, A. (2025). \newblock Mixing time of the proximal sampler in relative fisher information via strong data processing inequality. \newblock {\em arXiv preprint arXiv:2502.05623}. \bibitem[Xu et~al., 2024]{xu2024flow} Xu, C., Lee, J., Cheng, X., and Xie, Y. (2024). \newblock Flow-based distributionally robust optimization. \newblock {\em IEEE Journal on Selected Areas in Information Theory}, 5:62--77. \bibitem[Yu et~al., 2022]{yuFastDistributionallyRobust2022} Yu, Y., Lin, T., Mazumdar, E.~V., and Jordan, M. (2022). \newblock Fast {{Distributionally Robust Learning}} with {{Variance-Reduced Min-Max Optimization}}. \newblock In {\em Proceedings of {{The}} 25th {{International Conference}} on {{Artificial Intelligence}} and {{Statistics}}}, pages 1219--1250. PMLR. \bibitem[Zhao and Guan, 2018]{zhaoDatadrivenRiskaverseStochastic2018} Zhao, C. and Guan, Y. (2018). \newblock Data-driven risk-averse stochastic optimization with {{Wasserstein}} metric. \newblock {\em Operations Research Letters}, 46(2):262--267. \bibitem[Zhu et~al., 2021]{zhu2021kernel} Zhu, J.-J., Jitkrittum, W., Diehl, M., and Sch{\"o}lkopf, B. (2021). \newblock Kernel distributionally robust optimization: Generalized duality theorem and stochastic approximation. \newblock In {\em International Conference on Artificial Intelligence and Statistics}, pages 280--288. PMLR. \bibitem[Zhu and Xie, 2024]{zhu2024distributionally} Zhu, L. and Xie, Y. (2024). \newblock Distributionally robust optimization via iterative algorithms in continuous probability spaces. \newblock {\em arXiv preprint arXiv:2412.20556}. \end{thebibliography}