% SingleEq
@article{koncel2015parsing,
  title     = {Parsing algebraic word problems into equations},
  author    = {Koncel-Kedziorski, Rik and Hajishirzi, Hannaneh and Sabharwal, Ashish and Etzioni, Oren and Ang, Siena Dumas},
  journal   = {Transactions of the Association for Computational Linguistics},
  volume    = {3},
  pages     = {585--597},
  year      = {2015},
  publisher = {MIT Press One Rogers Street, Cambridge, MA 02142-1209, USA journals-info~…}
}

% SingleOp
@article{roy2015reasoning,
  title     = {Reasoning about quantities in natural language},
  author    = {Roy, Subhro and Vieira, Tim and Roth, Dan},
  journal   = {Transactions of the Association for Computational Linguistics},
  volume    = {3},
  pages     = {1--13},
  year      = {2015},
  publisher = {MIT Press One Rogers Street, Cambridge, MA 02142-1209, USA journals-info~…}
}

% MultiArith
@article{roy2016solving,
  title   = {Solving general arithmetic word problems},
  author  = {Roy, Subhro and Roth, Dan},
  journal = {arXiv preprint arXiv:1608.01413},
  year    = {2016}
}

% SVAMP
@article{patel2021nlp,
  title   = {Are NLP models really able to solve simple math word problems?},
  author  = {Patel, Arkil and Bhattamishra, Satwik and Goyal, Navin},
  journal = {arXiv preprint arXiv:2103.07191},
  year    = {2021}
}

% GSM8K
@article{cobbe2021gsm8k,
  title   = {Training Verifiers to Solve Math Word Problems},
  author  = {Cobbe, Karl and Kosaraju, Vineet and Bavarian, Mohammad and Chen, Mark and Jun, Heewoo and Kaiser, Lukasz and Plappert, Matthias and Tworek, Jerry and Hilton, Jacob and Nakano, Reiichiro and Hesse, Christopher and Schulman, John},
  journal = {arXiv preprint arXiv:2110.14168},
  year    = {2021}
}

% MMLU
@article{hendrycks2020measuring,
  title   = {Measuring massive multitask language understanding},
  author  = {Hendrycks, Dan and Burns, Collin and Basart, Steven and Zou, Andy and Mazeika, Mantas and Song, Dawn and Steinhardt, Jacob},
  journal = {arXiv preprint arXiv:2009.03300},
  year    = {2020}
}

% BigBench (Logic 3 Obj, Object Counting, Navigate)
@article{srivastava2022beyond,
  title   = {Beyond the imitation game: Quantifying and extrapolating the capabilities of language models},
  author  = {Srivastava, Aarohi and Rastogi, Abhinav and Rao, Abhishek and Shoeb, Abu Awal Md and Abid, Abubakar and Fisch, Adam and Brown, Adam R and Santoro, Adam and Gupta, Aditya and Garriga-Alonso, Adri{\`a} and others},
  journal = {arXiv preprint arXiv:2206.04615},
  year    = {2022}
}

% BigBenchHard (Logic 3 Obj, Object Counting, Navigate)
@article{suzgun2022challenging,
  title   = {Challenging big-bench tasks and whether chain-of-thought can solve them},
  author  = {Suzgun, Mirac and Scales, Nathan and Sch{\"a}rli, Nathanael and Gehrmann, Sebastian and Tay, Yi and Chung, Hyung Won and Chowdhery, Aakanksha and Le, Quoc V and Chi, Ed H and Zhou, Denny and others},
  journal = {arXiv preprint arXiv:2210.09261},
  year    = {2022}
}

% HotPotQA
@article{yang2018hotpotqa,
  title   = {HotpotQA: A dataset for diverse, explainable multi-hop question answering},
  author  = {Yang, Zhilin and Qi, Peng and Zhang, Saizheng and Bengio, Yoshua and Cohen, William W and Salakhutdinov, Ruslan and Manning, Christopher D},
  journal = {arXiv preprint arXiv:1809.09600},
  year    = {2018}
}

% SQuAD
@article{rajpurkar2016squad,
  title   = {Squad: 100,000+ questions for machine comprehension of text},
  author  = {Rajpurkar, Pranav and Zhang, Jian and Lopyrev, Konstantin and Liang, Percy},
  journal = {arXiv preprint arXiv:1606.05250},
  year    = {2016}
}

% SQuAD2.0
@article{rajpurkar2018know,
  title   = {Know what you don't know: Unanswerable questions for SQuAD},
  author  = {Rajpurkar, Pranav and Jia, Robin and Liang, Percy},
  journal = {arXiv preprint arXiv:1806.03822},
  year    = {2018}
}

% DROP
@article{dua2019drop,
  title   = {DROP: A reading comprehension benchmark requiring discrete reasoning over paragraphs},
  author  = {Dua, Dheeru and Wang, Yizhong and Dasigi, Pradeep and Stanovsky, Gabriel and Singh, Sameer and Gardner, Matt},
  journal = {arXiv preprint arXiv:1903.00161},
  year    = {2019}
}

% TabFact
@article{chen2019tabfact,
  title   = {Tabfact: A large-scale dataset for table-based fact verification},
  author  = {Chen, Wenhu and Wang, Hongmin and Chen, Jianshu and Zhang, Yunkai and Wang, Hong and Li, Shiyang and Zhou, Xiyou and Wang, William Yang},
  journal = {arXiv preprint arXiv:1909.02164},
  year    = {2019}
}

% Winograd Schema Challenge
@inproceedings{levesque2012winograd,
  title     = {The winograd schema challenge},
  author    = {Levesque, Hector and Davis, Ernest and Morgenstern, Leora},
  booktitle = {Thirteenth international conference on the principles of knowledge representation and reasoning},
  year      = {2012}
}

% VQA
@inproceedings{antol2015vqa,
  title={Vqa: Visual question answering},
  author={Antol, Stanislaw and Agrawal, Aishwarya and Lu, Jiasen and Mitchell, Margaret and Batra, Dhruv and Zitnick, C Lawrence and Parikh, Devi},
  booktitle={Proceedings of the IEEE international conference on computer vision},
  pages={2425--2433},
  year={2015}
}

%VQA v2.0
@inproceedings{goyal2017making,
  title={Making the v in vqa matter: Elevating the role of image understanding in visual question answering},
  author={Goyal, Yash and Khot, Tejas and Summers-Stay, Douglas and Batra, Dhruv and Parikh, Devi},
  booktitle={Proceedings of the IEEE conference on computer vision and pattern recognition},
  pages={6904--6913},
  year={2017}
}
