\begin{thebibliography}{82} \providecommand{\natexlab}[1]{#1} \providecommand{\url}[1]{\texttt{#1}} \expandafter\ifx\csname urlstyle\endcsname\relax \providecommand{\doi}[1]{doi: #1}\else \providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi \bibitem[Agrawal et~al.(2019{\natexlab{a}})Agrawal, Amos, Barratt, Boyd, Diamond, and Kolter]{aab+19} Agrawal, A., Amos, B., Barratt, S., Boyd, S., Diamond, S., and Kolter, J.~Z. \newblock Differentiable convex optimization layers. \newblock \emph{Advances in neural information processing systems}, 32, 2019{\natexlab{a}}. \bibitem[Agrawal et~al.(2019{\natexlab{b}})Agrawal, Barratt, Boyd, Busseti, and Moursi]{abb+19} Agrawal, A., Barratt, S., Boyd, S., Busseti, E., and Moursi, W.~M. \newblock Differentiating through a cone program. \newblock \emph{Journal of Applied and Numerical Optimization}, 1\penalty0 (2):\penalty0 107--115, 2019{\natexlab{b}}. \bibitem[Amos \& Kolter(2017)Amos and Kolter]{ak17} Amos, B. and Kolter, J.~Z. \newblock {OptNet}: Differentiable optimization as a layer in neural networks. \newblock In \emph{International conference on machine learning}, pp.\ 136--145. PMLR, 2017. \bibitem[Amos et~al.(2018)Amos, Jimenez, Sacks, Boots, and Kolter]{ajs+18} Amos, B., Jimenez, I., Sacks, J., Boots, B., and Kolter, J.~Z. \newblock Differentiable mpc for end-to-end planning and control. \newblock \emph{Advances in neural information processing systems}, 31, 2018. \bibitem[ApS(2025)]{mosek} ApS, M. \newblock \emph{The MOSEK Python Fusion API manual. Version 11.0.}, 2025. \newblock URL \url{https://docs.mosek.com/latest/pythonfusion/index.html}. \bibitem[Arbel \& Mairal(2022)Arbel and Mairal]{am22} Arbel, M. and Mairal, J. \newblock Non-convex bilevel games with critical point selection maps. \newblock \emph{Advances in Neural Information Processing Systems (NeurIPS)}, 35:\penalty0 8013--8026, 2022. \bibitem[Arjevani et~al.(2023)Arjevani, Carmon, Duchi, Foster, Srebro, and Woodworth]{acd+23} Arjevani, Y., Carmon, Y., Duchi, J.~C., Foster, D.~J., Srebro, N., and Woodworth, B. \newblock Lower bounds for non-convex stochastic optimization. \newblock \emph{Mathematical Programming}, 199\penalty0 (1):\penalty0 165--214, 2023. \bibitem[Bai et~al.(2021)Bai, Luo, Zhao, Wen, and Wang]{blz+2021} Bai, T., Luo, J., Zhao, J., Wen, B., and Wang, Q. \newblock Recent advances in adversarial training for adversarial robustness. \newblock In Zhou, Z.-H. (ed.), \emph{Proceedings of the Thirtieth International Joint Conference on Artificial Intelligence, {IJCAI-21}}, pp.\ 4312--4321. International Joint Conferences on Artificial Intelligence Organization, 8 2021. \newblock \doi{10.24963/ijcai.2021/591}. \newblock URL \url{https://doi.org/10.24963/ijcai.2021/591}. \newblock Survey Track. \bibitem[Bambade et~al.(2024)Bambade, Schramm, Taylor, and Carpentier]{bst+24} Bambade, A., Schramm, F., Taylor, A., and Carpentier, J. \newblock Leveraging augmented-lagrangian techniques for differentiating over infeasible quadratic programs in machine learning. \newblock In \emph{ICLR 2024-The Twelfth International Conference on Learning Representations}, 2024. \bibitem[Bao et~al.(2021)Bao, Wu, Li, Zhu, and Zhang]{bwl+21} Bao, F., Wu, G., Li, C., Zhu, J., and Zhang, B. \newblock Stability and generalization of bilevel programming in hyperparameter optimization. \newblock \emph{Advances in neural information processing systems}, 34:\penalty0 4529--4541, 2021. \bibitem[Bertrand et~al.(2022)Bertrand, Klopfenstein, Massias, Blondel, Vaiter, Gramfort, and Salmon]{bkm+22} Bertrand, Q., Klopfenstein, Q., Massias, M., Blondel, M., Vaiter, S., Gramfort, A., and Salmon, J. \newblock Implicit differentiation for fast hyperparameter selection in non-smooth convex learning. \newblock \emph{Journal of Machine Learning Research}, 23\penalty0 (149):\penalty0 1--43, 2022. \bibitem[Besan{\c{c}}on et~al.(2024)Besan{\c{c}}on, Dias~Garcia, Legat, and Sharma]{bdl+24} Besan{\c{c}}on, M., Dias~Garcia, J., Legat, B., and Sharma, A. \newblock Flexible differentiable optimization via model transformations. \newblock \emph{INFORMS Journal on Computing}, 36\penalty0 (2):\penalty0 456--478, 2024. \bibitem[Blondel et~al.(2022)Blondel, Berthet, Cuturi, Frostig, Hoyer, Llinares-L{\'o}pez, Pedregosa, and Vert]{bbc+22} Blondel, M., Berthet, Q., Cuturi, M., Frostig, R., Hoyer, S., Llinares-L{\'o}pez, F., Pedregosa, F., and Vert, J.-P. \newblock Efficient and modular implicit differentiation. \newblock \emph{Advances in neural information processing systems}, 35:\penalty0 5230--5242, 2022. \bibitem[Bolte et~al.(2023)Bolte, Pauwels, and Vaiter]{bpv23} Bolte, J., Pauwels, E., and Vaiter, S. \newblock One-step differentiation of iterative algorithms. \newblock \emph{Advances in Neural Information Processing Systems}, 36:\penalty0 77089--77103, 2023. \bibitem[Bracken \& McGill(1973)Bracken and McGill]{bm73} Bracken, J. and McGill, J.~T. \newblock Mathematical programs with optimization problems in the constraints. \newblock \emph{Operations research}, 21\penalty0 (1):\penalty0 37--44, 1973. \bibitem[Butler(2023)]{b23} Butler, A. \newblock {SCQPTH}: an efficient differentiable splitting method for convex quadratic programming. \newblock \emph{arXiv preprint arXiv:2308.08232}, 2023. \bibitem[Butler \& Kwon(2023)Butler and Kwon]{bk23} Butler, A. and Kwon, R.~H. \newblock Efficient differentiable quadratic programming layers: an admm approach. \newblock \emph{Computational Optimization and Applications}, 84\penalty0 (2):\penalty0 449--476, 2023. \bibitem[Caron et~al.(2025)Caron, Arnström, Bonagiri, Dechaume, Flowers, Heins, Ishikawa, Kenefake, Mazzamuto, Meoli, O'Donoghue, Oppenheimer, Pandala, Quiroz~Omaña, Rontsis, Shah, St-Jean, Vitucci, Wolfers, Yang, GitHub~user, MeindertHH, rimaddo, urob, shaoanlu, Khalil, Kozlov, Groudiev, Sousa~Pinto, Schwan, Budhiraja, and jkeust]{qpsolvers} Caron, S., Arnström, D., Bonagiri, S., Dechaume, A., Flowers, N., Heins, A., Ishikawa, T., Kenefake, D., Mazzamuto, G., Meoli, D., O'Donoghue, B., Oppenheimer, A.~A., Pandala, A., Quiroz~Omaña, J.~J., Rontsis, N., Shah, P., St-Jean, S., Vitucci, N., Wolfers, S., Yang, F., GitHub~user, b., MeindertHH, rimaddo, urob, shaoanlu, Khalil, A., Kozlov, L., Groudiev, A., Sousa~Pinto, J., Schwan, R., Budhiraja, R., and jkeust. \newblock {qpsolvers: Quadratic Programming Solvers in Python}, 2025. \newblock URL \url{https://github.com/qpsolvers/qpsolvers}. \bibitem[Chen et~al.(2024)Chen, Xu, and Zhang]{cxz24} Chen, L., Xu, J., and Zhang, J. \newblock On finding small hyper-gradients in bilevel optimization: Hardness results and improved analysis. \newblock In \emph{The Thirty Seventh Annual Conference on Learning Theory (COLT)}, pp.\ 947--980. PMLR, 2024. \bibitem[Chu et~al.(2024)Chu, Xu, Yao, and Zhang]{cxy+24} Chu, T., Xu, D., Yao, W., and Zhang, J. \newblock {SPABA}: A single-loop and probabilistic stochastic bilevel algorithm achieving optimal sample complexity. \newblock In \emph{Forty-first International Conference on Machine Learning (ICML)}, 2024. \newblock URL \url{https://openreview.net/forum?id=1YMjzz2g81}. \bibitem[Danilova et~al.(2022)Danilova, Dvurechensky, Gasnikov, Gorbunov, Guminov, Kamzolov, and Shibaev]{ddg+22} Danilova, M., Dvurechensky, P., Gasnikov, A., Gorbunov, E., Guminov, S., Kamzolov, D., and Shibaev, I. \newblock Recent theoretical advances in non-convex optimization. \newblock In \emph{High-Dimensional Optimization and Probability: With a View Towards Data Science}, pp.\ 79--163. Springer, 2022. \bibitem[Domke(2012)]{d12} Domke, J. \newblock Generic methods for optimization-based modeling. \newblock In \emph{Artificial Intelligence and Statistics}, pp.\ 318--326. PMLR, 2012. \bibitem[Dontchev \& Rockafellar(2009)Dontchev and Rockafellar]{dr09} Dontchev, A.~L. and Rockafellar, R.~T. \newblock \emph{Implicit functions and solution mappings}, volume 543. \newblock Springer, 2009. \bibitem[Donti et~al.(2017)Donti, Amos, and Kolter]{dak17} Donti, P., Amos, B., and Kolter, J.~Z. \newblock Task-based end-to-end model learning in stochastic optimization. \newblock \emph{Advances in neural information processing systems}, 30, 2017. \bibitem[Elsken et~al.(2019)Elsken, Metzen, and Hutter]{emh19} Elsken, T., Metzen, J.~H., and Hutter, F. \newblock Neural architecture search: A survey. \newblock \emph{Journal of Machine Learning Research}, 20\penalty0 (55):\penalty0 1--21, 2019. \bibitem[Feurer \& Hutter(2019)Feurer and Hutter]{fh19} Feurer, M. and Hutter, F. \newblock Hyperparameter optimization. \newblock \emph{Automated machine learning: Methods, systems, challenges}, pp.\ 3--33, 2019. \bibitem[Finn et~al.(2017)Finn, Abbeel, and Levine]{fal17} Finn, C., Abbeel, P., and Levine, S. \newblock Model-agnostic meta-learning for fast adaptation of deep networks. \newblock In \emph{International Conference on Machine Learning (ICML)}, pp.\ 1126--1135. PMLR, 2017. \bibitem[Franceschi et~al.(2018)Franceschi, Frasconi, Salzo, Grazzi, and Pontil]{ffs+18} Franceschi, L., Frasconi, P., Salzo, S., Grazzi, R., and Pontil, M. \newblock Bilevel programming for hyperparameter optimization and meta-learning. \newblock In \emph{International Conference on Machine Learning (ICML)}, pp.\ 1568--1577. PMLR, 2018. \bibitem[Ghadimi \& Wang(2018)Ghadimi and Wang]{gw18} Ghadimi, S. and Wang, M. \newblock Approximation methods for bilevel programming. \newblock \emph{arXiv preprint arXiv:1802.02246}, 2018. \bibitem[Goldstein(1977)]{g77} Goldstein, A.~A. \newblock Optimization of lipschitz continuous functions. \newblock \emph{Mathematical Programming}, 13\penalty0 (1):\penalty0 14--22, 1977. \bibitem[{Gurobi Optimization, LLC}(2025)]{gurobi} {Gurobi Optimization, LLC}. \newblock {Gurobi Optimizer Reference Manual}, 2025. \newblock URL \url{https://www.gurobi.com}. \bibitem[Healey et~al.(2025)Healey, Nobel, and Boyd]{hnb25} Healey, Q., Nobel, P., and Boyd, S. \newblock Differentiating through a quadratic cone program. \newblock \emph{arXiv preprint arXiv:2508.17522}, 2025. \bibitem[Holmes et~al.(2025)Holmes, D{\"u}mbgen, and Barfoot]{hdb25} Holmes, C., D{\"u}mbgen, F., and Barfoot, T.~D. \newblock {SDPRLayers}: Certifiable backpropagation through polynomial optimization problems in robotics. \newblock \emph{IEEE Transactions on Robotics}, 2025. \bibitem[Holstege et~al.(2024)Holstege, Wouters, van Giersbergen, and Diks]{hwg+2024} Holstege, F., Wouters, B., van Giersbergen, N., and Diks, C. \newblock Optimizing importance weighting in the presence of sub-population shifts. \newblock \emph{arXiv preprint arXiv:2410.14315}, 2024. \bibitem[Huang(2024)]{h24} Huang, F. \newblock Optimal hessian/jacobian-free nonconvex-pl bilevel optimization. \newblock In \emph{Forty-first International Conference on Machine Learning (ICML)}, 2024. \bibitem[Innes et~al.(2019)Innes, Edelman, Fischer, Rackauckas, Saba, Shah, and Tebbutt]{ief+19} Innes, M., Edelman, A., Fischer, K., Rackauckas, C., Saba, E., Shah, V.~B., and Tebbutt, W. \newblock A differentiable programming system to bridge machine learning and scientific computing. \newblock \emph{arXiv preprint arXiv:1907.07587}, 2019. \bibitem[Jiang et~al.(2024)Jiang, Xiao, Tenorio, Real-Rojas, Marques, and Chen]{jxt+24} Jiang, L., Xiao, Q., Tenorio, V.~M., Real-Rojas, F., Marques, A., and Chen, T. \newblock A primal-dual-assisted penalty approach to bilevel optimization with coupled constraints. \newblock In \emph{The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS)}, 2024. \newblock URL \url{https://openreview.net/forum?id=uZi7H5Ac0X}. \bibitem[Khanduri et~al.(2025)Khanduri, Tsaknakis, Zhang, Liu, and Hong]{khanduri2025doubly} Khanduri, P., Tsaknakis, I., Zhang, Y., Liu, S., and Hong, M. \newblock A doubly stochastically perturbed algorithm for linearly constrained bilevel optimization. \newblock \emph{arXiv preprint arXiv:2504.04545}, 2025. \bibitem[Kornowski et~al.(2024)Kornowski, Padmanabhan, Wang, Zhang, and Sra]{kpw+24} Kornowski, G., Padmanabhan, S., Wang, K., Zhang, Z., and Sra, S. \newblock First-order methods for linearly constrained bilevel optimization. \newblock In \emph{The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS)}, 2024. \bibitem[Krantz \& Parks(2002)Krantz and Parks]{sp02} Krantz, S.~G. and Parks, H.~R. \newblock \emph{The implicit function theorem: history, theory, and applications}. \newblock Springer Science \& Business Media, 2002. \bibitem[Kwon et~al.(2023)Kwon, Kwon, Wright, and Nowak]{kkw+23} Kwon, J., Kwon, D., Wright, S., and Nowak, R.~D. \newblock A fully first-order method for stochastic bilevel optimization. \newblock In \emph{International Conference on Machine Learning (ICML)}, pp.\ 18083--18113. PMLR, 2023. \bibitem[Kwon et~al.(2024{\natexlab{a}})Kwon, Kwon, and Lyu]{kkl24} Kwon, J., Kwon, D., and Lyu, H. \newblock On the complexity of first-order methods in stochastic bilevel optimization. \newblock In \emph{International Conference on Machine Learning (ICML)}, 2024{\natexlab{a}}. \bibitem[Kwon et~al.(2024{\natexlab{b}})Kwon, Kwon, Wright, and Nowak]{kkw+24} Kwon, J., Kwon, D., Wright, S., and Nowak, R.~D. \newblock On penalty methods for nonconvex bilevel optimization and first-order stochastic approximation. \newblock In \emph{International Conference on Learning Representations {(ICLR)}}, 2024{\natexlab{b}}. \newblock URL \url{https://openreview.net/forum?id=CvYBvgEUK9}. \bibitem[Lee et~al.(2019)Lee, Maji, Ravichandran, and Soatto]{lmr+19} Lee, K., Maji, S., Ravichandran, A., and Soatto, S. \newblock Meta-learning with differentiable convex optimization. \newblock In \emph{Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, pp.\ 10657--10665, 2019. \bibitem[Liu et~al.(2022)Liu, Ye, Wright, Stone, and Liu]{lyw+22} Liu, B., Ye, M., Wright, S., Stone, P., and Liu, Q. \newblock Bome! bilevel optimization made easy: A simple first-order approach. \newblock \emph{Advances in neural information processing systems}, 35:\penalty0 17248--17262, 2022. \bibitem[Liu et~al.(2021{\natexlab{a}})Liu, Liu, Yuan, Zeng, and Zhang]{lly+21} Liu, R., Liu, X., Yuan, X., Zeng, S., and Zhang, J. \newblock A value-function-based interior-point method for non-convex bi-level optimization. \newblock In \emph{International Conference on Machine Learning (ICML)}, pp.\ 6882--6892. PMLR, 2021{\natexlab{a}}. \bibitem[Liu et~al.(2021{\natexlab{b}})Liu, Liu, Zeng, and Zhang]{llz+21} Liu, R., Liu, Y., Zeng, S., and Zhang, J. \newblock Towards gradient-based bilevel optimization with non-convex followers and beyond. \newblock \emph{Advances in Neural Information Processing Systems (NeurIPS)}, 34:\penalty0 8662--8675, 2021{\natexlab{b}}. \bibitem[Liu et~al.(2024)Liu, Liu, Yao, Zeng, and Zhang]{lly+24} Liu, R., Liu, Z., Yao, W., Zeng, S., and Zhang, J. \newblock Moreau envelope for nonconvex bi-level optimization: A single-loop and hessian-free solution strategy. \newblock In \emph{Forty-first International Conference on Machine Learning}, 2024. \newblock URL \url{https://openreview.net/forum?id=rZD9hV0Bc4}. \bibitem[Lorraine et~al.(2020)Lorraine, Vicol, and Duvenaud]{lorraine2020optimizing} Lorraine, J., Vicol, P., and Duvenaud, D. \newblock Optimizing millions of hyperparameters by implicit differentiation. \newblock In \emph{International conference on artificial intelligence and statistics}, pp.\ 1540--1552. PMLR, 2020. \bibitem[Maclaurin et~al.(2015{\natexlab{a}})Maclaurin, Duvenaud, and Adams]{dda15} Maclaurin, D., Duvenaud, D., and Adams, R. \newblock Gradient-based hyperparameter optimization through reversible learning. \newblock In \emph{International conference on machine learning}, pp.\ 2113--2122. PMLR, 2015{\natexlab{a}}. \bibitem[Maclaurin et~al.(2015{\natexlab{b}})Maclaurin, Duvenaud, and Adams]{mda15} Maclaurin, D., Duvenaud, D., and Adams, R. \newblock Gradient-based hyperparameter optimization through reversible learning. \newblock In \emph{International Conference on Machine Learning (ICML)}, pp.\ 2113--2122. PMLR, 2015{\natexlab{b}}. \bibitem[Magoon et~al.(2025)Magoon, Yang, Aigerman, and Kovalsky]{mya+25} Magoon, C.~W., Yang, F., Aigerman, N., and Kovalsky, S.~Z. \newblock Differentiation through black-box quadratic programming solvers. \newblock In \emph{The Thirty-ninth Annual Conference on Neural Information Processing Systems}, 2025. \newblock URL \url{https://openreview.net/forum?id=DvwKWKG1Ul}. \bibitem[Mandi et~al.(2024)Mandi, Kotary, Berden, Mulamba, Bucarey, Guns, and Fioretto]{mkb+24} Mandi, J., Kotary, J., Berden, S., Mulamba, M., Bucarey, V., Guns, T., and Fioretto, F. \newblock Decision-focused learning: Foundations, state of the art, benchmark and future opportunities. \newblock \emph{Journal of Artificial Intelligence Research}, 80:\penalty0 1623--1701, 2024. \bibitem[O'Donoghue et~al.(2016)O'Donoghue, Chu, Parikh, and Boyd]{ocpb16} O'Donoghue, B., Chu, E., Parikh, N., and Boyd, S. \newblock Conic optimization via operator splitting and homogeneous self-dual embedding. \newblock \emph{Journal of Optimization Theory and Applications}, 169\penalty0 (3):\penalty0 1042--1068, June 2016. \newblock URL \url{http://stanford.edu/~boyd/papers/scs.html}. \bibitem[Pan et~al.(2024)Pan, Ye, Yang, Yang, Liu, Wang, and Bian]{pyy+24} Pan, J., Ye, Z., Yang, X., Yang, X., Liu, W., Wang, L., and Bian, J. \newblock {BPQP}: A differentiable convex optimization framework for efficient end-to-end learning. \newblock \emph{Advances in Neural Information Processing Systems}, 37:\penalty0 77468--77493, 2024. \bibitem[Paszke et~al.(2017)Paszke, Gross, Chintala, Chanan, Yang, DeVito, Lin, Desmaison, Antiga, and Lerer]{pgs+17} Paszke, A., Gross, S., Chintala, S., Chanan, G., Yang, E., DeVito, Z., Lin, Z., Desmaison, A., Antiga, L., and Lerer, A. \newblock Automatic differentiation in pytorch. \newblock In \emph{NIPS 2017 Workshop on Autodiff}, 2017. \bibitem[Paulus et~al.(2021)Paulus, Rol{\'\i}nek, Musil, Amos, and Martius]{prm+21} Paulus, A., Rol{\'\i}nek, M., Musil, V., Amos, B., and Martius, G. \newblock {CombOptNet}: Fit the right np-hard problem by learning integer programming constraints. \newblock In \emph{International Conference on Machine Learning}, pp.\ 8443--8453. PMLR, 2021. \bibitem[Paulus et~al.(2024)Paulus, Martius, and Musil]{pmm24} Paulus, A., Martius, G., and Musil, V. \newblock {LPGD}: A general framework for backpropagation through embedded optimization layers. \newblock In \emph{International Conference on Machine Learning}, pp.\ 39989--40014. PMLR, 2024. \bibitem[Petrulionyte et~al.(2024)Petrulionyte, Mairal, and Arbel]{pma24} Petrulionyte, I., Mairal, J., and Arbel, M. \newblock Functional bilevel optimization for machine learning. \newblock In \emph{The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS)}, 2024. \bibitem[Pineda et~al.(2022)Pineda, Fan, Monge, Venkataraman, Sodhi, Chen, Ortiz, DeTone, Wang, Anderson, et~al.]{pfm+22} Pineda, L., Fan, T., Monge, M., Venkataraman, S., Sodhi, P., Chen, R.~T., Ortiz, J., DeTone, D., Wang, A., Anderson, S., et~al. \newblock Theseus: A library for differentiable nonlinear optimization. \newblock \emph{Advances in Neural Information Processing Systems}, 35:\penalty0 3801--3818, 2022. \bibitem[Rajeswaran et~al.(2019)Rajeswaran, Finn, Kakade, and Levine]{rfk+19} Rajeswaran, A., Finn, C., Kakade, S.~M., and Levine, S. \newblock Meta-learning with implicit gradients. \newblock \emph{Advances in neural information processing systems}, 32, 2019. \bibitem[Razaviyayn et~al.(2020)Razaviyayn, Huang, Lu, Nouiehed, Sanjabi, and Hong]{rhl+20} Razaviyayn, M., Huang, T., Lu, S., Nouiehed, M., Sanjabi, M., and Hong, M. \newblock Nonconvex min-max optimization: Applications, challenges, and recent theoretical advances. \newblock \emph{IEEE Signal Processing Magazine}, 37\penalty0 (5):\penalty0 55--66, 2020. \bibitem[Ren et~al.(2023)Ren, Feng, Liu, Pan, Fu, Mai, and Yang]{rfl+23} Ren, J., Feng, X., Liu, B., Pan, X., Fu, Y., Mai, L., and Yang, Y. \newblock {TorchOpt}: An efficient library for differentiable optimization. \newblock \emph{Journal of Machine Learning Research}, 24\penalty0 (367):\penalty0 1--14, 2023. \bibitem[Rosemberg et~al.(2025)Rosemberg, Garcia, Pacaud, Parker, Legat, Sundar, Bent, and Van~Hentenryck]{rgp+25} Rosemberg, A.~W., Garcia, J.~D., Pacaud, F., Parker, R.~B., Legat, B., Sundar, K., Bent, R., and Van~Hentenryck, P. \newblock A general and streamlined differentiable optimization framework. \newblock \emph{arXiv preprint arXiv:2510.25986}, 2025. \bibitem[Saif et~al.(2024)Saif, Cui, Shen, Lu, Kingsbury, and Chen]{scs+24} Saif, A., Cui, X., Shen, H., Lu, S., Kingsbury, B., and Chen, T. \newblock Joint unsupervised and supervised training for automatic speech recognition via bilevel optimization. \newblock In \emph{ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)}, pp.\ 10931--10935. IEEE, 2024. \bibitem[Schaller \& Boyd(2025)Schaller and Boyd]{sb25} Schaller, M. and Boyd, S. \newblock Code generation for solving and differentiating through convex optimization problems. \newblock \emph{arXiv preprint arXiv:2504.14099}, 2025. \bibitem[Shen \& Chen(2023)Shen and Chen]{sc23} Shen, H. and Chen, T. \newblock On penalty-based bilevel gradient descent method. \newblock In \emph{International Conference on Machine Learning (ICML)}, pp.\ 30992--31015. PMLR, 2023. \bibitem[Shen et~al.(2024)Shen, Yang, and Chen]{syc24} Shen, H., Yang, Z., and Chen, T. \newblock Principled penalty-based methods for bilevel reinforcement learning and {RLHF}. \newblock In \emph{Forty-first International Conference on Machine Learning (ICML)}, 2024. \newblock URL \url{https://openreview.net/forum?id=Xb3IXEBYuw}. \bibitem[Stellato et~al.(2020)Stellato, Banjac, Goulart, Bemporad, and Boyd]{sbg+20} Stellato, B., Banjac, G., Goulart, P., Bemporad, A., and Boyd, S. \newblock {OSQP}: an operator splitting solver for quadratic programs. \newblock \emph{Mathematical Programming Computation}, 12\penalty0 (4):\penalty0 637--672, 2020. \newblock \doi{10.1007/s12532-020-00179-2}. \newblock URL \url{https://doi.org/10.1007/s12532-020-00179-2}. \bibitem[Stewart(1977)]{stewart1977perturbation} Stewart, G.~W. \newblock On the perturbation of pseudo-inverses, projections and linear least squares problems. \newblock \emph{SIAM review}, 19\penalty0 (4):\penalty0 634--662, 1977. \bibitem[Sun et~al.(2023)Sun, Shi, Wang, Tuan, Poor, and Tao]{ssw+23} Sun, H., Shi, Y., Wang, J., Tuan, H.~D., Poor, H.~V., and Tao, D. \newblock Alternating differentiation for optimization layers. \newblock In \emph{The Eleventh International Conference on Learning Representations}, 2023. \newblock URL \url{https://openreview.net/forum?id=KKBMz-EL4tD}. \bibitem[Tracy et~al.(2023)Tracy, Howell, and Manchester]{thm23} Tracy, K., Howell, T.~A., and Manchester, Z. \newblock Differentiable collision detection for a set of convex primitives. \newblock In \emph{2023 IEEE International Conference on Robotics and Automation (ICRA)}, pp.\ 3663--3670. IEEE, 2023. \bibitem[Von~Stackelberg et~al.(1953)Von~Stackelberg, Peacock, Schneider, and Hutchison]{vps+53} Von~Stackelberg, H., Peacock, A.~T., Schneider, E., and Hutchison, T. \newblock The theory of the market economy. \newblock \emph{Economica}, 20\penalty0 (80):\penalty0 384, 1953. \bibitem[Wilder et~al.(2019)Wilder, Dilkina, and Tambe]{wdt19} Wilder, B., Dilkina, B., and Tambe, M. \newblock Melding the data-decisions pipeline: Decision-focused learning for combinatorial optimization. \newblock In \emph{Proceedings of the AAAI conference on artificial intelligence}, volume~33, pp.\ 1658--1665, 2019. \bibitem[Xiao et~al.(2023)Xiao, Lu, and Chen]{xlc23} Xiao, Q., Lu, S., and Chen, T. \newblock An alternating optimization method for bilevel problems under the polyak-\l ojasiewicz condition. \newblock In \emph{Advances in Neural Information Processing Systems (NeurIPS)}, volume~36, pp.\ 63847--63873, 2023. \bibitem[Xue et~al.(2021)Xue, Wang, Yan, Hu, Yang, and Sun]{xwy+21} Xue, C., Wang, X., Yan, J., Hu, Y., Yang, X., and Sun, K. \newblock Rethinking bi-level optimization in neural architecture search: A gibbs sampling perspective. \newblock In \emph{Proceedings of the AAAI Conference on Artificial Intelligence}, volume~35, pp.\ 10551--10559, 2021. \bibitem[Yang et~al.(2021)Yang, Ji, and Liang]{yjl21} Yang, J., Ji, K., and Liang, Y. \newblock Provably faster algorithms for bilevel optimization. \newblock \emph{Advances in Neural Information Processing Systems}, 34:\penalty0 13670--13682, 2021. \bibitem[Yao et~al.(2024)Yao, Yin, Zeng, and Zhang]{yyz+24} Yao, W., Yin, H., Zeng, S., and Zhang, J. \newblock Overcoming lower-level constraints in bilevel optimization: A novel approach with regularized gap functions. \newblock \emph{arXiv preprint arXiv:2406.01992}, 2024. \bibitem[Ye \& Zhu(1995)Ye and Zhu]{yz95} Ye, J.~J. and Zhu, D. \newblock Optimality conditions for bilevel programming problems. \newblock \emph{Optimization}, 33\penalty0 (1):\penalty0 9--27, 1995. \bibitem[Zhang et~al.(2024)Zhang, Chen, Xu, and Zhang]{zcx+24} Zhang, H., Chen, L., Xu, J., and Zhang, J. \newblock Functionally constrained algorithm solves convex simple bilevel problem. \newblock In \emph{The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS)}, 2024. \newblock URL \url{https://openreview.net/forum?id=PAiGHJppam}. \bibitem[Zhang et~al.(2020{\natexlab{a}})Zhang, Lin, Jegelka, Sra, and Jadbabaie]{zhang2020complexity} Zhang, J., Lin, H., Jegelka, S., Sra, S., and Jadbabaie, A. \newblock Complexity of finding stationary points of nonconvex nonsmooth functions. \newblock In \emph{International Conference on Machine Learning}, pp.\ 11173--11182. PMLR, 2020{\natexlab{a}}. \bibitem[Zhang et~al.(2020{\natexlab{b}})Zhang, Lin, Jegelka, Sra, and Jadbabaie]{zlj+20} Zhang, J., Lin, H., Jegelka, S., Sra, S., and Jadbabaie, A. \newblock Complexity of finding stationary points of nonconvex nonsmooth functions. \newblock In \emph{International Conference on Machine Learning}, pp.\ 11173--11182. PMLR, 2020{\natexlab{b}}. \end{thebibliography}