
@article{lee2026,
	title = {OpenLearnLM Benchmark: A Unified Framework for Evaluating Knowledge, Skill, and Attitude in Educational Large Language Models},
	author = {Lee, Unggi and Lee, Sookbun and Choi, Heungsoo and Lee, Jinseo and Park, Haeun and Jeon, Younghoon and Cho, Sungmin and Kang, Minju and Koh, Junbo and Bae, Jiyeong and Nam, Minwoo and Eun, Juyeon and Jung, Yeonji and Jeong, Yeil},
	year = {2026},
	date = {2026},
	doi = {10.48550/ARXIV.2601.13882},
	url = {https://arxiv.org/abs/2601.13882}
}

@article{lee2026a,
	title = {Are Video Models Zero-Shot Learners and Reasoners in Education? EduVideoBench, A Knowledge-Skills-Attitude Benchmark for Educational Video Generation},
	author = {Lee, Unggi and Ahn, Hoyoung and Choi, Yoon and Eun, Seonmin and Jeong, Jahyun and Jin, Seonmin and Jung, Harmony and Kim, Hye Jin and Lee, Chaerin and Lee, Hyunji and Lee, Jeongjin and Lee, Soohwan and Oh, Young-Seok and Park, Jaehyeon and Ryu, Sun-ok and Shin, Sunyoung and Son, Yoorim and Park, Haeun and Jeong, Yeil},
	year = {2026},
	date = {2026},
	doi = {10.48550/ARXIV.2605.26918},
	url = {https://arxiv.org/abs/2605.26918}
}

@article{lin2026,
	title = {Towards responsible AI in education: A Delphi-AHP-based framework for evaluating educational large language models},
	author = {Lin, Pingrong and Deng, Qin and Zhou, Yanbian},
	year = {2026},
	month = {06},
	date = {2026-06},
	journal = {Computers and Education: Artificial Intelligence},
	pages = {100534},
	volume = {10},
	doi = {10.1016/j.caeai.2025.100534},
	url = {http://dx.doi.org/10.1016/j.caeai.2025.100534},
	langid = {en}
}

@article{wei2025,
	title = {ELMES: An Automated Framework for Evaluating Large Language Models in Educational Scenarios},
	author = {Wei, {Shou'ang} and Wang, Xinyun and Bi, Shuzhen and Chen, Jian and Li, Ruijia and Jiang, Bo and Lin, Xin and Zhang, Min and Song, Yu and Li, BingDong and Zhou, Aimin and Hao, Hao},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2507.22947},
	url = {https://arxiv.org/abs/2507.22947}
}

@article{quttainah2024,
	title = {Cost, Usability, Credibility, Fairness, Accountability, Transparency, and Explainability Framework for Safe and Effective Large Language Models in Medical Education: Narrative Review and Qualitative Study},
	author = {Quttainah, Majdi and Mishra, Vinaytosh and Madakam, Somayya and Lurie, Yotam and Mark, Shlomo},
	year = {2024},
	month = {04},
	date = {2024-04-23},
	journal = {JMIR AI},
	pages = {e51834},
	volume = {3},
	doi = {10.2196/51834},
	url = {http://dx.doi.org/10.2196/51834},
	langid = {en}
}

@article{nikiforova-ilieva2025,
	title = {Benchmarking and Evaluation Framework for Large Language Models in Education},
	author = {Nikiforova-Ilieva, Kalina and Georgiev, Tsvetozar},
	year = {2025},
	month = {11},
	date = {2025-11-26},
	journal = {2025 6th International Conference on Communications, Information, Electronic and Energy Systems (CIEES)},
	pages = {1--5},
	doi = {10.1109/ciees66347.2025.11300167},
	url = {http://dx.doi.org/10.1109/ciees66347.2025.11300167}
}

@article{shahzad2025,
	title = {A comprehensive review of large language models: issues and solutions in learning environments},
	author = {Shahzad, Tariq and Mazhar, Tehseen and Tariq, Muhammad Usman and Ahmad, Wasim and Ouahada, Khmaies and Hamam, Habib},
	year = {2025},
	month = {01},
	date = {2025-01-14},
	journal = {Discover Sustainability},
	volume = {6},
	number = {1},
	doi = {10.1007/s43621-025-00815-8},
	url = {http://dx.doi.org/10.1007/s43621-025-00815-8},
	langid = {en}
}

@article{lee2024,
	title = {The life cycle of large language models in education: A framework for understanding sources of bias},
	author = {Lee, Jinsook and Hicke, Yann and Yu, Renzhe and Brooks, Christopher and Kizilcec, {René F.}},
	year = {2024},
	month = {07},
	date = {2024-07-12},
	journal = {British Journal of Educational Technology},
	pages = {1982--2002},
	volume = {55},
	number = {5},
	doi = {10.1111/bjet.13505},
	url = {http://dx.doi.org/10.1111/bjet.13505},
	langid = {en}
}

@article{lelièvre2025,
	title = {Benchmarking the Pedagogical Knowledge of Large Language Models},
	author = {{Lelièvre}, Maxime and Waldock, Amy and Liu, Meng and Aspillaga, {Natalia Valdés} and Mackintosh, Alasdair and Portela, {María José Ogando} and Lee, Jared and Atherton, Paul and Ince, Robin A. A. and Garrod, Oliver G. B.},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2506.18710},
	url = {https://arxiv.org/abs/2506.18710}
}

@article{brown2024,
	title = {Enhancing Trust in LLMs: Algorithms for Comparing and Interpreting LLMs},
	author = {Brown, Nik Bear},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2406.01943},
	url = {https://arxiv.org/abs/2406.01943}
}

@article{alzahrani2024,
	title = {When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards},
	author = {Alzahrani, Norah and Alyahya, Hisham Abdullah and Alnumay, Yazeed and Alrashed, Sultan and Alsubaie, Shaykhah and Almushaykeh, Yusef and Mirza, Faisal and Alotaibi, Nouf and Altwairesh, Nora and Alowisheq, Areeb and Bari, M Saiful and Khan, Haidar},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2402.01781},
	url = {https://arxiv.org/abs/2402.01781}
}

@article{jiang2025,
	title = {EduGuardBench: A Holistic Benchmark for Evaluating the Pedagogical Fidelity and Adversarial Safety of LLMs as Simulated Teachers},
	author = {Jiang, Yilin and Zhang, Mingzi and Yin, Xuanyu and Jin, Sheng and Lu, Suyu and Ying, Zuocan and Yu, Zengyi and Kong, Xiangjie},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2511.06890},
	url = {https://arxiv.org/abs/2511.06890}
}

@article{liu2024,
	title = {MathBench: Evaluating the Theory and Application Proficiency of LLMs with a Hierarchical Mathematics Benchmark},
	author = {Liu, Hongwei and Zheng, Zilong and Qiao, Yuxuan and Duan, Haodong and Fei, Zhiwei and Zhou, Fengzhe and Zhang, Wenwei and Zhang, Songyang and Lin, Dahua and Chen, Kai},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2405.12209},
	url = {https://arxiv.org/abs/2405.12209}
}

@article{rooein2024,
	title = {Beyond Flesch-Kincaid: Prompt-based Metrics Improve Difficulty Classification of Educational Texts},
	author = {Rooein, Donya and Rottger, Paul and Shaitarova, Anastassia and Hovy, Dirk},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2405.09482},
	url = {https://arxiv.org/abs/2405.09482}
}

@article{jeon2026,
	title = {ISD-Agent-Bench: A Comprehensive Benchmark for Evaluating LLM-based Instructional Design Agents},
	author = {Jeon, YoungHoon and Kim, Suwan and Son, Haein and Lee, Sookbun and Jeong, Yeil and Lee, Unggi},
	year = {2026},
	date = {2026},
	doi = {10.48550/ARXIV.2602.10620},
	url = {https://arxiv.org/abs/2602.10620}
}

@article{zhou2025,
	title = {From Answers to Questions: EQGBench for Evaluating LLMs' Educational Question Generation},
	author = {Zhou, Chengliang and Wang, Mei and Zhang, Ting and Zhu, Qiannan and Li, Jian and Huang, Hua},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2508.10005},
	url = {https://arxiv.org/abs/2508.10005}
}

@article{balepur2026,
	title = {BenchMarker: An Education-Inspired Toolkit for Highlighting Flaws in Multiple-Choice Benchmarks},
	author = {Balepur, Nishant and Rajasekaran, Bhavya and Oh, Jane and Xie, Michael and Desai, Atrey and Gupta, Vipul and Moore, Steven James and Choi, Eunsol and Rudinger, Rachel and Boyd-Graber, Jordan Lee},
	year = {2026},
	date = {2026},
	doi = {10.48550/ARXIV.2602.06221},
	url = {https://arxiv.org/abs/2602.06221}
}

@article{moëll2025,
	title = {Swedish Medical LLM Benchmark: development and evaluation of a framework for assessing large language models in the Swedish medical domain},
	author = {{Moëll}, Birger and Farestam, Fabian and Beskow, Jonas},
	year = {2025},
	month = {07},
	date = {2025-07-11},
	journal = {Frontiers in Artificial Intelligence},
	volume = {8},
	doi = {10.3389/frai.2025.1557920},
	url = {http://dx.doi.org/10.3389/frai.2025.1557920}
}

@article{qian2026,
	title = {Benchmark{\textasciicircum}2: Systematic Evaluation of LLM Benchmarks},
	author = {Qian, Qi and Huang, Chengsong and Xu, Jingwen and Lv, Changze and Wu, Muling and Liu, Wenhao and Wang, Xiaohua and Wang, Zhenghua and Huang, Zisu and Tian, Muzhao and Xu, Jianhan and Hu, Kun and Wang, He-Da and Hu, Yao and Huang, Xuanjing and Zheng, Xiaoqing},
	year = {2026},
	date = {2026},
	doi = {10.48550/ARXIV.2601.03986},
	url = {https://arxiv.org/abs/2601.03986}
}

@article{jiao2025,
	title = {LLM ethics benchmark: a three-dimensional assessment system for evaluating moral reasoning in large language models},
	author = {Jiao, Junfeng and Afroogh, Saleh and Murali, Abhejay and Chen, Kevin and Atkinson, David and Dhurandhar, Amit},
	year = {2025},
	month = {10},
	date = {2025-10-05},
	journal = {Scientific Reports},
	volume = {15},
	number = {1},
	doi = {10.1038/s41598-025-18489-7},
	url = {http://dx.doi.org/10.1038/s41598-025-18489-7},
	langid = {en}
}

@article{hamna2026,
	title = {Building Benchmarks from the Ground Up: Community-Centered Evaluation of LLMs in Healthcare Chatbot Settings},
	author = {Hamna, Hamna and Bhat, Gayatri and Mukherjee, Sourabrata and Lalani, Faisal M. and Hadfield, Evan and Siddarth, Divya and Bali, Kalika and Sitaram, Sunayana},
	year = {2026},
	month = {04},
	date = {2026-04-13},
	journal = {Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems},
	pages = {1--19},
	doi = {10.1145/3772318.3791172},
	url = {http://dx.doi.org/10.1145/3772318.3791172}
}

@article{li2025,
	title = {OKBench: Democratizing LLM Evaluation with Fully Automated, On-Demand, Open Knowledge Benchmarking},
	author = {Li, Yanhong and Xu, Tianyang and Tang, Kenan and Livescu, Karen and McAllester, David and Zhou, Jiawei},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2511.08598},
	url = {https://arxiv.org/abs/2511.08598}
}

@article{singh2025,
	title = {The Leaderboard Illusion},
	author = {Singh, Shivalika and Nan, Yiyang and Wang, Alex and {D'Souza}, Daniel and Kapoor, Sayash and {Üstün}, Ahmet and Koyejo, Sanmi and Deng, Yuntian and Longpre, Shayne and Smith, Noah A. and Ermis, Beyza and Fadaee, Marzieh and Hooker, Sara},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2504.20879},
	url = {https://arxiv.org/abs/2504.20879}
}

@article{chen2025,
	title = {Harnessing Multiple Large Language Models: A Survey on LLM Ensemble},
	author = {Chen, Zhijun and Lu, Xiaodong and Li, Jingzheng and Chen, Pengpeng and Li, Zhuoran and Sun, Kai and Luo, Yuankai and Mao, Qianren and Li, Ming and Xiao, Likang and Yang, Dingqi and Huang, Xiao and Ban, Yikun and Sun, Hailong and Yu, Philip S.},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2502.18036},
	url = {https://arxiv.org/abs/2502.18036}
}

@article{zhang2025,
	title = {DataSciBench: An LLM Agent Benchmark for Data Science},
	author = {Zhang, Dan and Zhoubian, Sining and Cai, Min and Li, Fengzu and Yang, Lekang and Wang, Wei and Dong, Tianjiao and Hu, Ziniu and Tang, Jie and Yue, Yisong},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2502.13897},
	url = {https://arxiv.org/abs/2502.13897}
}

@article{arora2025,
	title = {HealthBench: Evaluating Large Language Models Towards Improved Human Health},
	author = {Arora, Rahul K. and Wei, Jason and Hicks, Rebecca Soskin and Bowman, Preston and {Quiñonero-Candela}, Joaquin and Tsimpourlas, Foivos and Sharman, Michael and Shah, Meghan and Vallone, Andrea and Beutel, Alex and Heidecke, Johannes and Singhal, Karan},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2505.08775},
	url = {https://arxiv.org/abs/2505.08775}
}

@article{lunardi2025,
	title = {On Robustness and Reliability of Benchmark-Based Evaluation of LLMs},
	author = {Lunardi, Riccardo and Della Mea, Vincenzo and Mizzaro, Stefano and Roitero, Kevin},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2509.04013},
	url = {https://arxiv.org/abs/2509.04013}
}

@article{potamitis2025,
	title = {ReasonBENCH: Benchmarking the (In)Stability of LLM Reasoning},
	author = {Potamitis, Nearchos and Ramani, Vansh and Arora, Har Ashish and Kuchhal, Dhairya and Klein, Lars and Arora, Akhil},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2512.07795},
	url = {https://arxiv.org/abs/2512.07795}
}

@article{wang2025,
	title = {Rethinking LLM Evaluation: Can We Evaluate LLMs with 200x Less Data?},
	author = {Wang, Shaobo and Wang, Cong and Fu, Wenjie and Min, Yue and Feng, Mingquan and Guan, Isabel and Hu, Xuming and He, Conghui and Wang, Cunxiang and Yang, Kexin and Ren, Xingzhang and Huang, Fei and Liu, Dayiheng and Zhang, Linfeng},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2510.10457},
	url = {https://arxiv.org/abs/2510.10457}
}

@article{shashidhar2025,
	title = {YourBench: Easy Custom Evaluation Sets for Everyone},
	author = {Shashidhar, Sumuk and Fourrier, {Clémentine} and Lozovskia, Alina and Wolf, Thomas and Tur, Gokhan and {Hakkani-Tür}, Dilek},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2504.01833},
	url = {https://arxiv.org/abs/2504.01833}
}

@article{feuer2025,
	title = {When Judgment Becomes Noise: How Design Failures in LLM Judge Benchmarks Silently Undermine Validity},
	author = {Feuer, Benjamin and Tseng, Chiung-Yi and Lathe, Astitwa Sarthak and Elachqar, Oussama and Dickerson, John P},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2509.20293},
	url = {https://arxiv.org/abs/2509.20293}
}

@article{banyas2025,
	title = {ConsistencyAI: A Benchmark to Assess LLMs' Factual Consistency When Responding to Different Demographic Groups},
	author = {Banyas, Peter and Sharma, Shristi and Simmons, Alistair and Vispute, Atharva},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2510.13852},
	url = {https://arxiv.org/abs/2510.13852}
}

@article{mohammadi2025,
	title = {Evaluation and Benchmarking of LLM Agents: A Survey},
	author = {Mohammadi, Mahmoud and Li, Yipeng and Lo, Jane and Yip, Wendy},
	year = {2025},
	month = {08},
	date = {2025-08-03},
	journal = {Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2},
	pages = {6129--6139},
	doi = {10.1145/3711896.3736570},
	url = {http://dx.doi.org/10.1145/3711896.3736570}
}

@article{zhu2026,
	title = {Benchmark Health Index: A Systematic Framework for Benchmarking the Benchmarks of LLMs},
	author = {Zhu, Longyuan and Hua, Hairan and Miao, Linlin and Zhao, Bing},
	year = {2026},
	date = {2026},
	doi = {10.48550/ARXIV.2602.11674},
	url = {https://arxiv.org/abs/2602.11674}
}

@article{kim2025,
	title = {BenchHub: A Unified Benchmark Suite for Holistic and Customizable LLM Evaluation},
	author = {Kim, Eunsu and Yoo, Haneul and Son, Guijin and Patel, Hitesh and Agarwal, Amit and Oh, Alice},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2506.00482},
	url = {https://arxiv.org/abs/2506.00482}
}

@article{devries2023,
	title = {DUMB: A Benchmark for Smart Evaluation of Dutch Models},
	author = {de Vries, Wietse and Wieling, Martijn and Nissim, Malvina},
	year = {2023},
	date = {2023},
	doi = {10.48550/ARXIV.2305.13026},
	url = {https://arxiv.org/abs/2305.13026}
}

@article{vanroy2023,
	title = {Language Resources for Dutch Large Language Modelling},
	author = {Vanroy, Bram},
	year = {2023},
	date = {2023},
	doi = {10.48550/ARXIV.2312.12852},
	url = {https://arxiv.org/abs/2312.12852}
}

@article{vanroy2024,
	title = {Fietje: An open, efficient LLM for Dutch},
	author = {Vanroy, Bram},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2412.15450},
	url = {https://arxiv.org/abs/2412.15450}
}

@article{noels2024,
	title = {A Dutch Financial Large Language Model},
	author = {Noels, Sander and De Blaere, Jorne and De Bie, Tijl},
	year = {2024},
	month = {11},
	date = {2024-11-14},
	journal = {Proceedings of the 5th ACM International Conference on AI in Finance},
	pages = {283--291},
	doi = {10.1145/3677052.3698628},
	url = {http://dx.doi.org/10.1145/3677052.3698628}
}

@article{thellmann2024,
	title = {Towards Multilingual LLM Evaluation for European Languages},
	author = {Thellmann, Klaudia and Stadler, Bernhard and Fromm, Michael and Buschhoff, Jasper Schulze and Jude, Alex and Barth, Fabio and Leveling, Johannes and Flores-Herr, Nicolas and {Köhler}, Joachim and {Jäkel}, {René} and Ali, Mehdi},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2410.08928},
	url = {https://arxiv.org/abs/2410.08928}
}

@article{barth2025,
	title = {Multilingual European Language Models: Benchmarking Approaches and Challenges},
	author = {Barth, Fabio and Rehm, Georg},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2502.12895},
	url = {https://arxiv.org/abs/2502.12895}
}

@article{vintar2025,
	title = {Charting the European LLM Benchmarking Landscape: A New Taxonomy and a Set of Best Practices},
	author = {Vintar, {{\v{S}}pela} and {Punger{\v{s}}ek}, Taja Kuzman and Brglez, Mojca and {Ljube{\v{s}}i{\'{c}}}, Nikola},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2510.24450},
	url = {https://arxiv.org/abs/2510.24450}
}

@article{martins2024,
	title = {EuroLLM: Multilingual Language Models for Europe},
	author = {Martins, Pedro Henrique and Fernandes, Patrick and Alves, {João} and Guerreiro, Nuno M. and Rei, Ricardo and Alves, Duarte M. and Pombal, {José} and Farajian, Amin and Faysse, Manuel and Klimaszewski, Mateusz and Colombo, Pierre and Haddow, Barry and de Souza, {José G. C.} and Birch, Alexandra and Martins, {André F. T.}},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2409.16235},
	url = {https://arxiv.org/abs/2409.16235}
}

@article{son2024,
	title = {MM-Eval: A Multilingual Meta-Evaluation Benchmark for LLM-as-a-Judge and Reward Models},
	author = {Son, Guijin and Yoon, Dongkeun and Suk, Juyoung and Aula-Blasco, Javier and Aslan, Mano and Kim, Vu Trong and Islam, Shayekh Bin and {Prats-Cristià}, Jaume and {Tormo-Bañuelos}, {Lucía} and Kim, Seungone},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2410.17578},
	url = {https://arxiv.org/abs/2410.17578}
}

@article{chang2024,
	title = {A Survey on Evaluation of Large Language Models},
	author = {Chang, Yupeng and Wang, Xu and Wang, Jindong and Wu, Yuan and Yang, Linyi and Zhu, Kaijie and Chen, Hao and Yi, Xiaoyuan and Wang, Cunxiang and Wang, Yidong and Ye, Wei and Zhang, Yue and Chang, Yi and Yu, Philip S. and Yang, Qiang and Xie, Xing},
	year = {2024},
	month = {03},
	date = {2024-03-29},
	journal = {ACM Transactions on Intelligent Systems and Technology},
	pages = {1--45},
	volume = {15},
	number = {3},
	doi = {10.1145/3641289},
	url = {http://dx.doi.org/10.1145/3641289},
	langid = {en}
}

@article{ghosh2024,
	title = {ONEBench to Test Them All: Sample-Level Benchmarking Over Open-Ended Capabilities},
	author = {Ghosh, Adhiraj and Dziadzio, Sebastian and Prabhu, Ameya and Udandarao, Vishaal and Albanie, Samuel and Bethge, Matthias},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2412.06745},
	url = {https://arxiv.org/abs/2412.06745}
}

@article{gao2025,
	title = {FlowerTune: A Cross-Domain Benchmark for Federated Fine-Tuning of Large Language Models},
	author = {Gao, Yan and Scamarcia, Massimo Roberto and Fernandez-Marques, Javier and Naseri, Mohammad and Ng, Chong Shen and Stripelis, Dimitris and Li, Zexi and Shen, Tao and Bai, Jiamu and Chen, Daoyuan and Zhang, Zikai and Hu, Rui and Song, InSeo and KangYoon, Lee and Jia, Hong and Dang, Ting and Wang, Junyan and Liu, Zheyuan and Beutel, Daniel Janes and Lyu, Lingjuan and Lane, Nicholas D.},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2506.02961},
	url = {https://arxiv.org/abs/2506.02961}
}

@article{jegham2025,
	title = {How Hungry is AI? Benchmarking Energy, Water, and Carbon Footprint of LLM Inference},
	author = {Jegham, Nidhal and Abdelatti, Marwan and Koh, Chan Young and Elmoubarki, Lassad and Hendawi, Abdeltawab},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2505.09598},
	url = {https://arxiv.org/abs/2505.09598}
}

@article{mehditabar2025,
	title = {Smart but Costly? Benchmarking LLMs on Functional Accuracy and Energy Efficiency},
	author = {Mehditabar, Mohammadjavad and Rajput, Saurabhsingh and Mastropaolo, Antonio and Sharma, Tushar},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2511.07698},
	url = {https://arxiv.org/abs/2511.07698}
}

@article{yuan2025,
	title = {EfficientLLM: Efficiency in Large Language Models},
	author = {Yuan, Zhengqing and Sun, Weixiang and Liu, Yixin and Zhou, Huichi and Zhou, Rong and Li, Yiyang and Zhang, Zheyuan and Song, Wei and Huang, Yue and Jia, Haolong and Murugesan, Keerthiram and Wang, Yu and He, Lifang and Gao, Jianfeng and Sun, Lichao and Ye, Yanfang},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2505.13840},
	url = {https://arxiv.org/abs/2505.13840}
}

@article{husom2025,
	title = {Sustainable LLM Inference for Edge AI: Evaluating Quantized LLMs for Energy Efficiency, Output Accuracy, and Inference Latency},
	author = {Husom, Erik Johannes and Goknil, Arda and Astekin, Merve and Shar, Lwin Khin and {KÃ¥sen}, Andre and Sen, Sagar and Mithassel, Benedikt Andreas and Soylu, Ahmet},
	year = {2025},
	month = {11},
	date = {2025-11-18},
	journal = {ACM Transactions on Internet of Things},
	pages = {1--35},
	volume = {6},
	number = {4},
	doi = {10.1145/3767742},
	url = {http://dx.doi.org/10.1145/3767742},
	langid = {en}
}

@article{wu2025,
	title = {Unveiling Environmental Impacts of Large Language Model Serving: A Functional Unit View},
	author = {Wu, Yanran and Hua, Inez and Ding, Yi},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2502.11256},
	url = {https://arxiv.org/abs/2502.11256}
}

@article{dauner2025,
	title = {Energy costs of communicating with AI},
	author = {Dauner, Maximilian and Socher, Gudrun},
	year = {2025},
	month = {06},
	date = {2025-06-19},
	journal = {Frontiers in Communication},
	volume = {10},
	doi = {10.3389/fcomm.2025.1572947},
	url = {http://dx.doi.org/10.3389/fcomm.2025.1572947}
}

@article{vijay2025,
	title = {The Hidden Costs of Translation Accuracy: Distillation, Quantization, and Environmental Impact},
	author = {Vijay, Dhaathri and Vadapalli, Anandaswarup},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2509.23990},
	url = {https://arxiv.org/abs/2509.23990}
}

@article{huang2025a,
	title = {Evaluating the Quality of AI-Generated Digital Educational Resources for University Teaching and Learning},
	author = {Huang, Qian and Lv, Chunlan and Lu, Li and Tu, Shuang},
	year = {2025},
	month = {03},
	date = {2025-03-03},
	journal = {Systems},
	pages = {174},
	volume = {13},
	number = {3},
	doi = {10.3390/systems13030174},
	url = {http://dx.doi.org/10.3390/systems13030174},
	langid = {en}
}

@article{budler2025,
	title = {A Brief Review on Benchmarking for Large Language Models Evaluation in Healthcare},
	author = {Budler, Leona Cilar and Chen, Hongyu and Chen, Aokun and Topaz, Maxim and Tam, Wilson and Bian, Jiang and Stiglic, Gregor},
	year = {2025},
	month = {04},
	date = {2025-04-09},
	journal = {WIREs Data Mining and Knowledge Discovery},
	volume = {15},
	number = {2},
	doi = {10.1002/widm.70010},
	url = {http://dx.doi.org/10.1002/widm.70010},
	langid = {en}
}

@article{cámara2024,
	title = {Towards Standardized benchmarks of LLMs in software modeling tasks: a conceptual framework},
	author = {{Cámara}, Javier and {Burgueño}, Lola and Troya, Javier},
	year = {2024},
	month = {09},
	date = {2024-09-03},
	journal = {Software and Systems Modeling},
	pages = {1309--1318},
	volume = {23},
	number = {6},
	doi = {10.1007/s10270-024-01206-9},
	url = {http://dx.doi.org/10.1007/s10270-024-01206-9},
	langid = {en}
}

@article{mcintosh2026,
	title = {Inadequacies of Large Language Model Benchmarks in the Era of Generative Artificial Intelligence},
	author = {McIntosh, Timothy R. and Susnjak, Teo and Arachchilage, Nalin and Liu, Tong and Xu, Dan and Watters, Paul and Halgamuge, Malka N.},
	year = {2026},
	month = {01},
	date = {2026-01},
	journal = {IEEE Transactions on Artificial Intelligence},
	pages = {22--39},
	volume = {7},
	number = {1},
	doi = {10.1109/tai.2025.3569516},
	url = {http://dx.doi.org/10.1109/tai.2025.3569516}
}

@article{guo2023,
	title = {Evaluating Large Language Models: A Comprehensive Survey},
	author = {Guo, Zishan and Jin, Renren and Liu, Chuang and Huang, Yufei and Shi, Dan and {Supryadi} and Yu, Linhao and Liu, Yan and Li, Jiaxuan and Xiong, Bojian and Xiong, Deyi},
	year = {2023},
	date = {2023},
	doi = {10.48550/ARXIV.2310.19736},
	url = {https://arxiv.org/abs/2310.19736}
}

@article{xu2024,
	title = {Benchmark Data Contamination of Large Language Models: A Survey},
	author = {Xu, Cheng and Guan, Shuhao and Greene, Derek and Kechadi, M-Tahar},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2406.04244},
	url = {https://arxiv.org/abs/2406.04244}
}

@article{shool2025,
	title = {A systematic review of large language model (LLM) evaluations in clinical medicine},
	author = {Shool, Sina and Adimi, Sara and Saboori Amleshi, Reza and Bitaraf, Ehsan and Golpira, Reza and Tara, Mahmood},
	year = {2025},
	month = {03},
	date = {2025-03-07},
	journal = {BMC Medical Informatics and Decision Making},
	volume = {25},
	number = {1},
	doi = {10.1186/s12911-025-02954-4},
	url = {http://dx.doi.org/10.1186/s12911-025-02954-4},
	langid = {en}
}

@article{banerjee2024,
	title = {The Vulnerability of Language Model Benchmarks: Do They Accurately Reflect True LLM Performance?},
	author = {Banerjee, Sourav and Agarwal, Ayushi and Singh, Eishkaran},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2412.03597},
	url = {https://arxiv.org/abs/2412.03597}
}

@article{polo2024,
	title = {tinyBenchmarks: evaluating LLMs with fewer examples},
	author = {Polo, Felipe Maia and Weber, Lucas and Choshen, Leshem and Sun, Yuekai and Xu, Gongjun and Yurochkin, Mikhail},
	year = {2024},
	date = {2024},
	doi = {10.48550/ARXIV.2402.14992},
	url = {https://arxiv.org/abs/2402.14992}
}

@article{definelicht2023,
	title = {Integrating Large Language Models into Higher Education: Guidelines for Effective Implementation},
	author = {de Fine Licht, Karl},
	year = {2023},
	month = {08},
	date = {2023-08-11},
	journal = {IS4SI Summit 2023},
	pages = {65},
	doi = {10.3390/cmsf2023008065},
	url = {http://dx.doi.org/10.3390/cmsf2023008065}
}

@article{pomerenke2025,
	title = {The AI Language Proficiency Monitor -- Tracking the Progress of LLMs on Multilingual Benchmarks},
	author = {Pomerenke, David and Nothnagel, Jonas and Ostermann, Simon},
	year = {2025},
	date = {2025},
	doi = {10.48550/ARXIV.2507.08538},
	url = {https://arxiv.org/abs/2507.08538}
}
