@article{vaswani2017attention, title={Attention is All you Need}, author={Vaswani, Ashish and Shazeer, Noam and Parmar, Niki and Uszkoreit, Jakob and Jones, Llion and Gomez, Aidan N and Kaiser, {\L}ukasz and Polosukhin, Illia}, journal={Advances in neural information processing systems}, volume={30}, year={2017}, url={https://arxiv.org/abs/1706.03762}, eprint={1706.03762} } @article{devlin2018bert, title={BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding}, author={Devlin, Jacob and Chang, Ming-Wei and Lee, Kenton and Toutanova, Kristina}, journal={arXiv preprint arXiv:1810.04805}, year={2018}, eprint={1810.04805} } @inproceedings{brown2020language, title={Language models are few-shot learners}, author={Brown, Tom and Mann, Benjamin and Ryder, Nick and Subbiah, Melanie and Kaplan, Jared D and Dhariwal, Prafulla and Neelakantan, Arvind and Shyam, Pranav and Sastry, Girish and Askell, Amanda and others}, booktitle={Advances in Neural Information Processing Systems}, volume={33}, pages={1877--1901}, year={2020}, url={https://arxiv.org/abs/2005.14165} } @article{fake2023paper, title={This is a Fake Paper That Does Not Exist}, author={Nobody, John and Doesnotexist, Jane}, journal={Fake Journal}, year={2023}, url={https://example.com/nonexistent} } @article{Briani2007, author = {Briani, M. and Natalini, R. and Russo, G.}, title = {{I}mplicit–{E}xplicit numerical schemes for jump–diffusion processes}, journal = {Calcolo}, vol = {44}, pages = {33--57}, year = {2007}, doi={10.1007/s10092-007-0128-0}, }