Buckets:
| @article{mercer1909functions, | |
| title={Functions of positive and negative type, and their connection with the theory of integral equations}, | |
| author={Mercer, James}, | |
| journal={Phil. Trans. R. Soc. Lond. A}, | |
| volume={209}, | |
| pages={415--446}, | |
| year={1909} | |
| } | |
| @inproceedings{williams2000effect, | |
| title={The effect of the input density distribution on kernel-based classifiers}, | |
| author={Williams, Christopher K. I. and Seeger, Matthias}, | |
| booktitle={Proceedings of the 17th International Conference on Machine Learning (ICML)}, | |
| volume={1}, | |
| pages={1159--1166}, | |
| year={2000}, | |
| publisher={Morgan Kaufmann} | |
| } | |
| @inproceedings{rahimi2007random, | |
| title={Random features for large-scale kernel machines}, | |
| author={Rahimi, Ali and Recht, Benjamin}, | |
| booktitle={Advances in neural information processing systems}, | |
| year={2007} | |
| } | |
| @misc{hinton2015distillingknowledgeneuralnetwork, | |
| title={Distilling the Knowledge in a Neural Network}, | |
| author={Geoffrey Hinton and Oriol Vinyals and Jeff Dean}, | |
| year={2015}, | |
| eprint={1503.02531}, | |
| archivePrefix={arXiv}, | |
| primaryClass={stat.ML}, | |
| url={https://arxiv.org/abs/1503.02531}, | |
| } | |
| @misc{dieuleveut2016nonparametricstochasticapproximationlarge, | |
| title={Non-parametric Stochastic Approximation with Large Step sizes}, | |
| author={Aymeric Dieuleveut and Francis Bach}, | |
| year={2016}, | |
| eprint={1408.0361}, | |
| archivePrefix={arXiv}, | |
| primaryClass={math.ST}, | |
| url={https://arxiv.org/abs/1408.0361}, | |
| } | |
| @article{basri2019convergence, | |
| title={The convergence rate of neural networks for learned functions of different frequencies}, | |
| author={Basri, Ronen and Jacobs, David and Kasten, Yoni and Kritchman, Shira}, | |
| journal={Advances in Neural Information Processing Systems}, | |
| volume={32}, | |
| year={2019} | |
| } | |
| @inproceedings{jacot2018neural, | |
| title={Neural tangent kernel: Convergence and generalization in neural networks}, | |
| author={Jacot, Arthur and Gabriel, Franck and Hongler, Cl{\'e}ment}, | |
| booktitle={Advances in neural information processing systems}, | |
| year={2018} | |
| } | |
| @article{cao2019generalization, | |
| title={Generalization bounds of stochastic gradient descent for wide and deep neural networks}, | |
| author={Cao, Yuan and Gu, Quanquan}, | |
| journal={Advances in Neural Information Processing Systems}, | |
| volume={32}, | |
| year={2019} | |
| } | |
| @incollection{Bengio+chapter2007, | |
| author = {Bengio, Yoshua and LeCun, Yann}, | |
| booktitle = {Large Scale Kernel Machines}, | |
| publisher = {MIT Press}, | |
| title = {Scaling Learning Algorithms Towards {AI}}, | |
| year = {2007} | |
| } | |
| @inproceedings{Jain2017, | |
| doi = {10.4230/LIPICS.FSTTCS.2017.2}, | |
| url = {https://drops.dagstuhl.de/entities/document/10.4230/LIPIcs.FSTTCS.2017.2}, | |
| author = {Jain, Prateek and Kakade, Sham M. and Kidambi, Rahul and Netrapalli, Praneeth and Pillutla, Venkata Krishna and Sidford, Aaron}, | |
| keywords = {Stochastic Gradient Descent, Minimax Optimality, Least Squares Regression}, | |
| language = {en}, | |
| title = {A Markov Chain Theory Approach to Characterizing the Minimax Optimality of Stochastic Gradient Descent (for Least Squares)}, | |
| publisher = {Schloss Dagstuhl – Leibniz-Zentrum für Informatik}, | |
| year = {2018}, | |
| copyright = {Creative Commons Attribution 3.0 Unported license} | |
| } | |
| @article{bach2017equivalence, | |
| author = {Bach, Francis}, | |
| title = {On the Equivalence between Kernel Quadrature Rules and Random Feature Expansions}, | |
| journal = {Journal of Machine Learning Research}, | |
| year = {2017}, | |
| volume = {18}, | |
| number = {21}, | |
| pages = {1--38}, | |
| url = {http://jmlr.org/papers/v18/15-178.html} | |
| } | |
| @inproceedings{jain2018accelerating, | |
| title={Accelerating stochastic gradient descent for least squares regression}, | |
| author={Jain, Prateek and Kakade, Sham M and Kidambi, Rahul and Netrapalli, Praneeth and Sidford, Aaron}, | |
| booktitle={Conference on Learning Theory}, | |
| year={2018}, | |
| } | |
| @misc{li2018measuringintrinsicdimensionobjective, | |
| title={Measuring the Intrinsic Dimension of Objective Landscapes}, | |
| author={Chunyuan Li and Heerad Farkhoor and Rosanne Liu and Jason Yosinski}, | |
| year={2018}, | |
| eprint={1804.08838}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/1804.08838}, | |
| } | |
| @misc{furlanello2018bornneuralnetworks, | |
| title={Born Again Neural Networks}, | |
| author={Tommaso Furlanello and Zachary C. Lipton and Michael Tschannen and Laurent Itti and Anima Anandkumar}, | |
| year={2018}, | |
| eprint={1805.04770}, | |
| archivePrefix={arXiv}, | |
| primaryClass={stat.ML}, | |
| url={https://arxiv.org/abs/1805.04770}, | |
| } | |
| @inproceedings{bietti2019inductive, | |
| title={On the inductive bias of neural tangent kernels}, | |
| author={Bietti, Alberto and Mairal, Julien}, | |
| booktitle={Advances in Neural Information Processing Systems}, | |
| volume={32}, | |
| year={2019} | |
| } | |
| @inproceedings{ge2019step, | |
| title={The step decay schedule: A near optimal, geometrically decaying learning rate procedure for least squares}, | |
| author={Ge, Rong and Kakade, Sham M and Kidambi, Rahul and Netrapalli, Praneeth}, | |
| booktitle={Neural Information Processing Systems}, | |
| year={2019} | |
| } | |
| @misc{jacot2020neuraltangentkernelconvergence, | |
| title={Neural Tangent Kernel: Convergence and Generalization in Neural Networks}, | |
| author={Arthur Jacot and Franck Gabriel and Clément Hongler}, | |
| year={2020}, | |
| eprint={1806.07572}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/1806.07572}, | |
| } | |
| @misc{aghajanyan2020intrinsicdimensionalityexplainseffectiveness, | |
| title={Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning}, | |
| author={Armen Aghajanyan and Luke Zettlemoyer and Sonal Gupta}, | |
| year={2020}, | |
| eprint={2012.13255}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2012.13255}, | |
| } | |
| @inproceedings{bordelon2020spectrum, | |
| title={Spectrum dependent learning curves in kernel regression and wide neural networks}, | |
| author={Bordelon, Blake and Canatar, Abdulkadir and Pehlevan, Cengiz}, | |
| booktitle={Proceedings of the 37th International Conference on Machine Learning (ICML)}, | |
| pages={1024--1034}, | |
| year={2020}, | |
| organization={PMLR} | |
| } | |
| @article{Bartlett_2020, | |
| title={Benign overfitting in linear regression}, | |
| volume={117}, | |
| ISSN={1091-6490}, | |
| url={http://dx.doi.org/10.1073/pnas.1907378117}, | |
| DOI={10.1073/pnas.1907378117}, | |
| number={48}, | |
| journal={Proceedings of the National Academy of Sciences}, | |
| publisher={Proceedings of the National Academy of Sciences}, | |
| author={Bartlett, Peter L. and Long, Philip M. and Lugosi, Gábor and Tsigler, Alexander}, | |
| year={2020}, | |
| month=apr, pages={30063–30070} } | |
| @article{canatar2021spectral, | |
| title={Spectral bias and task-model alignment explain generalization in kernel regression and infinitely wide neural networks}, | |
| author={Canatar, Abdulkadir and Bordelon, Blake and Pehlevan, Cengiz}, | |
| journal={Nature Communications}, | |
| volume={12}, | |
| number={1}, | |
| pages={2914}, | |
| year={2021}, | |
| publisher={Nature Publishing Group} | |
| } | |
| @misc{pan2022eigencurveoptimallearningrate, | |
| title={Eigencurve: Optimal Learning Rate Schedule for SGD on Quadratic Objectives with Skewed Hessian Spectrums}, | |
| author={Rui Pan and Haishan Ye and Tong Zhang}, | |
| year={2022}, | |
| eprint={2110.14109}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2110.14109}, | |
| } | |
| @inproceedings{menon2021statistical, | |
| title={Statistical perspective on distillation}, | |
| author={Menon, Aditya Krishna and Rawat, Ankit Singh and Reddi, Sashank J and Kim, Seungyeon and Kumar, Sanjiv}, | |
| booktitle={International Conference on Machine Learning}, | |
| pages={7651--7662}, | |
| year={2021}, | |
| organization={PMLR} | |
| } | |
| @misc{stanton2021doesknowledgedistillationreally, | |
| title={Does Knowledge Distillation Really Work?}, | |
| author={Samuel Stanton and Pavel Izmailov and Polina Kirichenko and Alexander A. Alemi and Andrew Gordon Wilson}, | |
| year={2021}, | |
| eprint={2106.05945}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2106.05945}, | |
| } | |
| @inproceedings{zou2021benign, | |
| title={Benign overfitting of constant-stepsize {SGD} for linear regression}, | |
| author={Zou, Difan and Wu, Jingfeng and Braverman, Vladimir and Gu, Quanquan and Kakade, Sham}, | |
| booktitle={Conference on Learning Theory}, | |
| year={2021} | |
| } | |
| @inproceedings{wu2022last, | |
| title={Last iterate risk bounds of {SGD} with decaying stepsize for overparameterized linear regression}, | |
| author={Wu, Jingfeng and Zou, Difan and Braverman, Vladimir and Gu, Quanquan and Kakade, Sham}, | |
| booktitle={International Conference on Machine Learning}, | |
| year={2022} | |
| } | |
| @misc{wei2022toyrandommatrixmodels, | |
| title={More Than a Toy: Random Matrix Models Predict How Real-World Neural Representations Generalize}, | |
| author={Alexander Wei and Wei Hu and Jacob Steinhardt}, | |
| year={2022}, | |
| eprint={2203.06176}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2203.06176}, | |
| } | |
| @misc{tsigler2022benignoverfittingridgeregression, | |
| title={Benign overfitting in ridge regression}, | |
| author={A. Tsigler and P. L. Bartlett}, | |
| year={2022}, | |
| eprint={2009.14286}, | |
| archivePrefix={arXiv}, | |
| primaryClass={math.ST}, | |
| url={https://arxiv.org/abs/2009.14286}, | |
| } | |
| @misc{mobahi2020selfdistillationamplifiesregularizationhilbert, | |
| title={Self-Distillation Amplifies Regularization in Hilbert Space}, | |
| author={Hossein Mobahi and Mehrdad Farajtabar and Peter L. Bartlett}, | |
| year={2020}, | |
| eprint={2002.05715}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2002.05715}, | |
| } | |
| @misc{li2023riskboundsacceleratedsgd, | |
| title={Risk Bounds of Accelerated SGD for Overparameterized Linear Regression}, | |
| author={Xuheng Li and Yihe Deng and Jingfeng Wu and Dongruo Zhou and Quanquan Gu}, | |
| year={2023}, | |
| eprint={2311.14222}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2311.14222}, | |
| } | |
| @misc{malladi2023kernelbasedviewlanguagemodel, | |
| title={A Kernel-Based View of Language Model Fine-Tuning}, | |
| author={Sadhika Malladi and Alexander Wettig and Dingli Yu and Danqi Chen and Sanjeev Arora}, | |
| year={2023}, | |
| eprint={2210.05643}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2210.05643}, | |
| } | |
| @article{Burns2023WeaktoStrongGE, | |
| title={Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision}, | |
| author={Collin Burns and Pavel Izmailov and Jan Hendrik Kirchner and Bowen Baker and Leo Gao and Leopold Aschenbrenner and Yining Chen and Adrien Ecoffet and Manas R. Joglekar and Jan Leike and Ilya Sutskever and Jeff Wu and OpenAI}, | |
| journal={ArXiv}, | |
| year={2023}, | |
| volume={abs/2312.09390}, | |
| url={https://api.semanticscholar.org/CorpusID:266312608} | |
| } | |
| @inproceedings{Agrawal2022alphaReQA, | |
| title={$\alpha$-ReQ : Assessing Representation Quality in Self-Supervised Learning by measuring eigenspectrum decay}, | |
| author={Kumar Krishna Agrawal and Arnab Kumar Mondal and Arna Ghosh and Blake Aaron Richards}, | |
| booktitle={Neural Information Processing Systems}, | |
| year={2022}, | |
| url={https://api.semanticscholar.org/CorpusID:258509089} | |
| } | |
| @misc{garrido2023rankmeassessingdownstreamperformance, | |
| title={RankMe: Assessing the downstream performance of pretrained self-supervised representations by their rank}, | |
| author={Quentin Garrido and Randall Balestriero and Laurent Najman and Yann Lecun}, | |
| year={2023}, | |
| eprint={2210.02885}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2210.02885}, | |
| } | |
| @misc{allenzhu2023understandingensembleknowledgedistillation, | |
| title={Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning}, | |
| author={Zeyuan Allen-Zhu and Yuanzhi Li}, | |
| year={2023}, | |
| eprint={2012.09816}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2012.09816}, | |
| } | |
| @misc{zhang2024optimalityacceleratedsgdhighdimensional, | |
| title={The Optimality of (Accelerated) SGD for High-Dimensional Quadratic Optimization}, | |
| author={Haihan Zhang and Yuanshi Liu and Qianwen Chen and Cong Fang}, | |
| year={2024}, | |
| eprint={2409.09745}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2409.09745}, | |
| } | |
| @misc{nagarajan2024studentteacherdeviationsdistillationdoes, | |
| title={On student-teacher deviations in distillation: does it pay to disobey?}, | |
| author={Vaishnavh Nagarajan and Aditya Krishna Menon and Srinadh Bhojanapalli and Hossein Mobahi and Sanjiv Kumar}, | |
| year={2024}, | |
| eprint={2301.12923}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2301.12923}, | |
| } | |
| @article{Wu2024ProvableWG, | |
| title={Provable Weak-to-Strong Generalization via Benign Overfitting}, | |
| author={David X. Wu and Anant Sahai}, | |
| journal={ArXiv}, | |
| year={2024}, | |
| volume={abs/2410.04638}, | |
| url={https://api.semanticscholar.org/CorpusID:273185855} | |
| } | |
| @misc{charikar2024quantifyinggainweaktostronggeneralization, | |
| title={Quantifying the Gain in Weak-to-Strong Generalization}, | |
| author={Moses Charikar and Chirag Pabbaraju and Kirankumar Shiragur}, | |
| year={2024}, | |
| eprint={2405.15116}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2405.15116}, | |
| } | |
| @article{Atanasov2024ScalingAR, | |
| title={Scaling and renormalization in high-dimensional regression}, | |
| author={Alexander Atanasov and Jacob A. Zavatone-Veth and Cengiz Pehlevan}, | |
| journal={ArXiv}, | |
| year={2024}, | |
| volume={abs/2405.00592}, | |
| url={https://api.semanticscholar.org/CorpusID:269484262} | |
| } | |
| @misc{wang2023selfinstructaligninglanguagemodels, | |
| title={Self-Instruct: Aligning Language Models with Self-Generated Instructions}, | |
| author={Yizhong Wang and Yeganeh Kordi and Swaroop Mishra and Alisa Liu and Noah A. Smith and Daniel Khashabi and Hannaneh Hajishirzi}, | |
| year={2023}, | |
| eprint={2212.10560}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.CL}, | |
| url={https://arxiv.org/abs/2212.10560}, | |
| } | |
| @misc{abdin2024phi3technicalreporthighly, | |
| title={Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone}, | |
| author={Marah Abdin and Jyoti Aneja and Hany Awadalla and Ahmed Awadallah and Ammar Ahmad Awan and Nguyen Bach and Amit Bahree and Arash Bakhtiari and Jianmin Bao and Harkirat Behl and Alon Benhaim and Misha Bilenko and Johan Bjorck and Sébastien Bubeck and Martin Cai and Qin Cai and Vishrav Chaudhary and Dong Chen and Dongdong Chen and Weizhu Chen and Yen-Chun Chen and Yi-Ling Chen and Hao Cheng and Parul Chopra and Xiyang Dai and Matthew Dixon and Ronen Eldan and Victor Fragoso and Jianfeng Gao and Mei Gao and Min Gao and Amit Garg and Allie Del Giorno and Abhishek Goswami and Suriya Gunasekar and Emman Haider and Junheng Hao and Russell J. Hewett and Wenxiang Hu and Jamie Huynh and Dan Iter and Sam Ade Jacobs and Mojan Javaheripi and Xin Jin and Nikos Karampatziakis and Piero Kauffmann and Mahoud Khademi and Dongwoo Kim and Young Jin Kim and Lev Kurilenko and James R. Lee and Yin Tat Lee and Yuanzhi Li and Yunsheng Li and Chen Liang and Lars Liden and Xihui Lin and Zeqi Lin and Ce Liu and Liyuan Liu and Mengchen Liu and Weishung Liu and Xiaodong Liu and Chong Luo and Piyush Madan and Ali Mahmoudzadeh and David Majercak and Matt Mazzola and Caio César Teodoro Mendes and Arindam Mitra and Hardik Modi and Anh Nguyen and Brandon Norick and Barun Patra and Daniel Perez-Becker and Thomas Portet and Reid Pryzant and Heyang Qin and Marko Radmilac and Liliang Ren and Gustavo de Rosa and Corby Rosset and Sambudha Roy and Olatunji Ruwase and Olli Saarikivi and Amin Saied and Adil Salim and Michael Santacroce and Shital Shah and Ning Shang and Hiteshi Sharma and Yelong Shen and Swadheen Shukla and Xia Song and Masahiro Tanaka and Andrea Tupini and Praneetha Vaddamanu and Chunyu Wang and Guanhua Wang and Lijuan Wang and Shuohang Wang and Xin Wang and Yu Wang and Rachel Ward and Wen Wen and Philipp Witte and Haiping Wu and Xiaoxia Wu and Michael Wyatt and Bin Xiao and Can Xu and Jiahang Xu and Weijian Xu and Jilong Xue and Sonali Yadav and Fan Yang and Jianwei Yang and Yifan Yang and Ziyi Yang and Donghan Yu and Lu Yuan and Chenruidong Zhang and Cyril Zhang and Jianwen Zhang and Li Lyna Zhang and Yi Zhang and Yue Zhang and Yunan Zhang and Xiren Zhou}, | |
| year={2024}, | |
| eprint={2404.14219}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.CL}, | |
| url={https://arxiv.org/abs/2404.14219}, | |
| } | |
| @misc{dosovitskiy2021imageworth16x16words, | |
| title={An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale}, | |
| author={Alexey Dosovitskiy and Lucas Beyer and Alexander Kolesnikov and Dirk Weissenborn and Xiaohua Zhai and Thomas Unterthiner and Mostafa Dehghani and Matthias Minderer and Georg Heigold and Sylvain Gelly and Jakob Uszkoreit and Neil Houlsby}, | |
| year={2021}, | |
| eprint={2010.11929}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.CV}, | |
| url={https://arxiv.org/abs/2010.11929}, | |
| } | |
| @article{Guo_2025, | |
| title={DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning}, | |
| volume={645}, | |
| ISSN={1476-4687}, | |
| url={http://dx.doi.org/10.1038/s41586-025-09422-z}, | |
| DOI={10.1038/s41586-025-09422-z}, | |
| number={8081}, | |
| journal={Nature}, | |
| publisher={Springer Science and Business Media LLC}, | |
| author={Guo, Daya and Yang, Dejian and Zhang, Haowei and Song, Junxiao and Wang, Peiyi and Zhu, Qihao and Xu, Runxin and Zhang, Ruoyu and Ma, Shirong and Bi, Xiao and Zhang, Xiaokang and Yu, Xingkai and Wu, Yu and Wu, Z. F. and Gou, Zhibin and Shao, Zhihong and Li, Zhuoshu and Gao, Ziyi and Liu, Aixin and Xue, Bing and Wang, Bingxuan and Wu, Bochao and Feng, Bei and Lu, Chengda and Zhao, Chenggang and Deng, Chengqi and Ruan, Chong and Dai, Damai and Chen, Deli and Ji, Dongjie and Li, Erhang and Lin, Fangyun and Dai, Fucong and Luo, Fuli and Hao, Guangbo and Chen, Guanting and Li, Guowei and Zhang, H. and Xu, Hanwei and Ding, Honghui and Gao, Huazuo and Qu, Hui and Li, Hui and Guo, Jianzhong and Li, Jiashi and Chen, Jingchang and Yuan, Jingyang and Tu, Jinhao and Qiu, Junjie and Li, Junlong and Cai, J. L. and Ni, Jiaqi and Liang, Jian and Chen, Jin and Dong, Kai and Hu, Kai and You, Kaichao and Gao, Kaige and Guan, Kang and Huang, Kexin and Yu, Kuai and Wang, Lean and Zhang, Lecong and Zhao, Liang and Wang, Litong and Zhang, Liyue and Xu, Lei and Xia, Leyi and Zhang, Mingchuan and Zhang, Minghua and Tang, Minghui and Zhou, Mingxu and Li, Meng and Wang, Miaojun and Li, Mingming and Tian, Ning and Huang, Panpan and Zhang, Peng and Wang, Qiancheng and Chen, Qinyu and Du, Qiushi and Ge, Ruiqi and Zhang, Ruisong and Pan, Ruizhe and Wang, Runji and Chen, R. J. and Jin, R. L. and Chen, Ruyi and Lu, Shanghao and Zhou, Shangyan and Chen, Shanhuang and Ye, Shengfeng and Wang, Shiyu and Yu, Shuiping and Zhou, Shunfeng and Pan, Shuting and Li, S. S. and Zhou, Shuang and Wu, Shaoqing and Yun, Tao and Pei, Tian and Sun, Tianyu and Wang, T. and Zeng, Wangding and Liu, Wen and Liang, Wenfeng and Gao, Wenjun and Yu, Wenqin and Zhang, Wentao and Xiao, W. L. and An, Wei and Liu, Xiaodong and Wang, Xiaohan and Chen, Xiaokang and Nie, Xiaotao and Cheng, Xin and Liu, Xin and Xie, Xin and Liu, Xingchao and Yang, Xinyu and Li, Xinyuan and Su, Xuecheng and Lin, Xuheng and Li, X. Q. and Jin, Xiangyue and Shen, Xiaojin and Chen, Xiaosha and Sun, Xiaowen and Wang, Xiaoxiang and Song, Xinnan and Zhou, Xinyi and Wang, Xianzu and Shan, Xinxia and Li, Y. K. and Wang, Y. Q. and Wei, Y. X. and Zhang, Yang and Xu, Yanhong and Li, Yao and Zhao, Yao and Sun, Yaofeng and Wang, Yaohui and Yu, Yi and Zhang, Yichao and Shi, Yifan and Xiong, Yiliang and He, Ying and Piao, Yishi and Wang, Yisong and Tan, Yixuan and Ma, Yiyang and Liu, Yiyuan and Guo, Yongqiang and Ou, Yuan and Wang, Yuduan and Gong, Yue and Zou, Yuheng and He, Yujia and Xiong, Yunfan and Luo, Yuxiang and You, Yuxiang and Liu, Yuxuan and Zhou, Yuyang and Zhu, Y. X. and Huang, Yanping and Li, Yaohui and Zheng, Yi and Zhu, Yuchen and Ma, Yunxian and Tang, Ying and Zha, Yukun and Yan, Yuting and Ren, Z. Z. and Ren, Zehui and Sha, Zhangli and Fu, Zhe and Xu, Zhean and Xie, Zhenda and Zhang, Zhengyan and Hao, Zhewen and Ma, Zhicheng and Yan, Zhigang and Wu, Zhiyu and Gu, Zihui and Zhu, Zijia and Liu, Zijun and Li, Zilin and Xie, Ziwei and Song, Ziyang and Pan, Zizheng and Huang, Zhen and Xu, Zhipeng and Zhang, Zhongyu and Zhang, Zhen}, | |
| year={2025}, | |
| month=sep, pages={633–638} } | |
| @misc{lin2025scalinglawslinearregression, | |
| title={Scaling Laws in Linear Regression: Compute, Parameters, and Data}, | |
| author={Licong Lin and Jingfeng Wu and Sham M. Kakade and Peter L. Bartlett and Jason D. Lee}, | |
| year={2025}, | |
| eprint={2406.08466}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2406.08466}, | |
| } | |
| @misc{zhang2025learningcurvesstochasticgradient, | |
| title={Learning Curves of Stochastic Gradient Descent in Kernel Regression}, | |
| author={Haihan Zhang and Weicheng Lin and Yuanshi Liu and Cong Fang}, | |
| year={2025}, | |
| eprint={2505.22048}, | |
| archivePrefix={arXiv}, | |
| primaryClass={stat.ML}, | |
| url={https://arxiv.org/abs/2505.22048}, | |
| } | |
| @misc{dong2025discrepanciesvirtueweaktostronggeneralization, | |
| title={Discrepancies are Virtue: Weak-to-Strong Generalization through Lens of Intrinsic Dimension}, | |
| author={Yijun Dong and Yicheng Li and Yunai Li and Jason D. Lee and Qi Lei}, | |
| year={2025}, | |
| eprint={2502.05075}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2502.05075}, | |
| } | |
| @misc{ildiz2025highdimensionalanalysisknowledgedistillation, | |
| title={High-dimensional Analysis of Knowledge Distillation: Weak-to-Strong Generalization and Scaling Laws}, | |
| author={M. Emrullah Ildiz and Halil Alperen Gozeten and Ege Onur Taga and Marco Mondelli and Samet Oymak}, | |
| year={2025}, | |
| eprint={2410.18837}, | |
| archivePrefix={arXiv}, | |
| primaryClass={stat.ML}, | |
| url={https://arxiv.org/abs/2410.18837}, | |
| } | |
| @misc{medvedev2025weaktostronggeneralizationrandomfeature, | |
| title={Weak-to-Strong Generalization Even in Random Feature Networks, Provably}, | |
| author={Marko Medvedev and Kaifeng Lyu and Dingli Yu and Sanjeev Arora and Zhiyuan Li and Nathan Srebro}, | |
| year={2025}, | |
| eprint={2503.02877}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2503.02877}, | |
| } | |
| @misc{mulgund2025relatingmisfitgainweaktostrong, | |
| title={Relating Misfit to Gain in Weak-to-Strong Generalization Beyond the Squared Loss}, | |
| author={Abhijeet Mulgund and Chirag Pabbaraju}, | |
| year={2025}, | |
| eprint={2501.19105}, | |
| archivePrefix={arXiv}, | |
| primaryClass={cs.LG}, | |
| url={https://arxiv.org/abs/2501.19105}, | |
| } | |
| @misc{moniri2025mechanismsweaktostronggeneralizationtheoretical, | |
| title={On the Mechanisms of Weak-to-Strong Generalization: A Theoretical Perspective}, | |
| author={Behrad Moniri and Hamed Hassani}, | |
| year={2025}, | |
| eprint={2505.18346}, | |
| archivePrefix={arXiv}, | |
| primaryClass={stat.ML}, | |
| url={https://arxiv.org/abs/2505.18346}, | |
| } | |
| @inproceedings{zhang2017age, | |
| title={Age progression/regression by conditional adversarial autoencoder}, | |
| author={Zhang, Zhifei and Song, Yang and Qi, Hairong}, | |
| booktitle={Proceedings of the IEEE conference on computer vision and pattern recognition}, | |
| pages={5810--5818}, | |
| year={2017} | |
| } | |
| @software{torchvision2016, | |
| title = {TorchVision: PyTorch's Computer Vision library}, | |
| author = {TorchVision maintainers and contributors}, | |
| year = 2016, | |
| journal = {GitHub repository}, | |
| publisher = {GitHub}, | |
| howpublished = {\url{https://github.com/pytorch/vision}} | |
| } | |
| @inproceedings{he2016deep, | |
| title={Deep residual learning for image recognition}, | |
| author={He, Kaiming and Zhang, Xiangyu and Ren, Shaoqing and Sun, Jian}, | |
| booktitle={Proceedings of the IEEE conference on computer vision and pattern recognition}, | |
| pages={770--778}, | |
| year={2016} | |
| } | |
| @inproceedings{radford2021learning, | |
| title={Learning transferable visual models from natural language supervision}, | |
| author={Radford, Alec and Kim, Jong Wook and Hallacy, Chris and Ramesh, Aditya and Goh, Gabriel and Agarwal, Sandhini and Sastry, Girish and Askell, Amanda and Mishkin, Pamela and Clark, Jack and others}, | |
| booktitle={International conference on machine learning}, | |
| pages={8748--8763}, | |
| year={2021}, | |
| organization={PmLR} | |
| } | |
| @inproceedings{wolf-etal-2020-transformers, | |
| title = "Transformers: State-of-the-Art Natural Language Processing", | |
| author = "Thomas Wolf and Lysandre Debut and Victor Sanh and Julien Chaumond and Clement Delangue and Anthony Moi and Pierric Cistac and Tim Rault and Rémi Louf and Morgan Funtowicz and Joe Davison and Sam Shleifer and Patrick von Platen and Clara Ma and Yacine Jernite and Julien Plu and Canwen Xu and Teven Le Scao and Sylvain Gugger and Mariama Drame and Quentin Lhoest and Alexander M. Rush", | |
| booktitle = "Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations", | |
| month = oct, | |
| year = "2020", | |
| address = "Online", | |
| publisher = "Association for Computational Linguistics", | |
| url = "https://www.aclweb.org/anthology/2020.emnlp-demos.6", | |
| pages = "38--45" | |
| } | |
| @book{gonzalez2008digital, | |
| title={Digital Image Processing}, | |
| author={Gonzalez, R.C. and Woods, R.E.}, | |
| isbn={9780131687288}, | |
| lccn={2009289249}, | |
| url={https://books.google.com.hk/books?id=8uGOnjRGEzoC}, | |
| year={2008}, | |
| publisher={Prentice Hall} | |
| } | |
| @inproceedings{caron2021emerging, | |
| title={Emerging properties in self-supervised vision transformers}, | |
| author={Caron, Mathilde and Touvron, Hugo and Misra, Ishan and J{\'e}gou, Herv{\'e} and Mairal, Julien and Bojanowski, Piotr and Joulin, Armand}, | |
| booktitle={Proceedings of the IEEE/CVF international conference on computer vision}, | |
| pages={9650--9660}, | |
| year={2021} | |
| } | |
| @inproceedings{radosavovic2020designing, | |
| title={Designing network design spaces}, | |
| author={Radosavovic, Ilija and Kosaraju, Raj Prateek and Girshick, Ross and He, Kaiming and Doll{\'a}r, Piotr}, | |
| booktitle={Proceedings of the IEEE/CVF conference on computer vision and pattern recognition}, | |
| pages={10428--10436}, | |
| year={2020} | |
| } | |
| @inproceedings{tan2019efficientnet, | |
| title={Efficientnet: Rethinking model scaling for convolutional neural networks}, | |
| author={Tan, Mingxing and Le, Quoc}, | |
| booktitle={International conference on machine learning}, | |
| pages={6105--6114}, | |
| year={2019}, | |
| organization={PMLR} | |
| } | |
| @article{krizhevsky2012imagenet, | |
| title={Imagenet classification with deep convolutional neural networks}, | |
| author={Krizhevsky, Alex and Sutskever, Ilya and Hinton, Geoffrey E}, | |
| journal={Advances in neural information processing systems}, | |
| volume={25}, | |
| year={2012} | |
| } | |
| @inproceedings{deng2009imagenet, | |
| title={Imagenet: A large-scale hierarchical image database}, | |
| author={Deng, Jia and Dong, Wei and Socher, Richard and Li, Li-Jia and Li, Kai and Fei-Fei, Li}, | |
| booktitle={2009 IEEE conference on computer vision and pattern recognition}, | |
| pages={248--255}, | |
| year={2009}, | |
| organization={Ieee} | |
| } |
Xet Storage Details
- Size:
- 27.6 kB
- Xet hash:
- d99c0cc413e54edcdba62580882589f99a2e1a234967b4c83d675aeb9b386d85
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.