{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T01:51:46Z","timestamp":1765504306080,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":71,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,10]]},"DOI":"10.1145\/3746252.3761133","type":"proceedings-article","created":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T00:29:28Z","timestamp":1762561768000},"page":"1497-1507","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Pedagogy-R1: Pedagogical Large Reasoning Model and Well-balanced Educational Benchmark"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0883-4128","authenticated-orcid":false,"given":"Unggi","family":"Lee","sequence":"first","affiliation":[{"name":"Chosun University, Gwangju, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5431-0973","authenticated-orcid":false,"given":"Jaeyong","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4430-7518","authenticated-orcid":false,"given":"Jiyeong","family":"Bae","sequence":"additional","affiliation":[{"name":"Enuma, Inc., Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4524-7197","authenticated-orcid":false,"given":"Yeil","family":"Jeong","sequence":"additional","affiliation":[{"name":"Indiana University, Bloomington, Indiana, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3463-0091","authenticated-orcid":false,"given":"Junbo","family":"Koh","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7844-9412","authenticated-orcid":false,"given":"Gyeonggeon 'Boaz'","family":"Lee","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5457-3706","authenticated-orcid":false,"given":"Gunho","family":"Lee","sequence":"additional","affiliation":[{"name":"Enuma, Inc., Berkeley, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9394-1650","authenticated-orcid":false,"given":"Taekyung","family":"Ahn","sequence":"additional","affiliation":[{"name":"Enuma, Inc., Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0555-8591","authenticated-orcid":false,"given":"Hyeoncheol","family":"Kim","sequence":"additional","affiliation":[{"name":"Korea University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,10]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"DBE-KT22: A Knowledge Tracing Dataset Based on Online Student Evaluation. arXiv preprint arXiv:2208.12651","author":"Abdelrahman Ghodai","year":"2022","unstructured":"Ghodai Abdelrahman, Sherif Abdelfattah, Qing Wang, and Yu Lin. 2022. DBE-KT22: A Knowledge Tracing Dataset Based on Online Student Evaluation. arXiv preprint arXiv:2208.12651 (2022)."},{"key":"e_1_3_2_1_2_1","unstructured":"AI-For-Education.org. 2025. The Pedagogy Benchmark. https:\/\/benchmarks.ai-for-education.org\/#moreinfo-about-the-benchmark"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of 12th International Conference on Learning Analytics and Knowledge (LAK22)","author":"AlZoubi Dana","year":"2022","unstructured":"Dana AlZoubi. 2022. From data to actions: Unfolding instructors' sense-making and reflective practice with classroom analytics. In Proceedings of 12th International Conference on Learning Analytics and Knowledge (LAK22)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.61969\/jai.1337500"},{"volume-title":"Dimensions of thinking and cognitive instruction","author":"Borko Hilda","key":"e_1_3_2_1_5_1","unstructured":"Hilda Borko and Richard J Shavelson. 2013. Teacher decision making 1. In Dimensions of thinking and cognitive instruction. Routledge, 311-346."},{"key":"e_1_3_2_1_6_1","volume-title":"Towards reasoning era: A survey of long chain-of-thought for reasoning large language models. arXiv preprint arXiv:2503.09567","author":"Chen Qiguang","year":"2025","unstructured":"Qiguang Chen, Libo Qin, Jinhao Liu, Dengyun Peng, Jiannan Guan, Peng Wang, Mengkang Hu, Yuhang Zhou, Te Gao, and Wanxiang Che. 2025b. Towards reasoning era: A survey of long chain-of-thought for reasoning large language models. arXiv preprint arXiv:2503.09567 (2025)."},{"key":"e_1_3_2_1_7_1","volume-title":"SEAL: Steerable Reasoning Calibration of Large Language Models for Free. arXiv preprint arXiv:2504.07986","author":"Chen Runjin","year":"2025","unstructured":"Runjin Chen, Zhenyu Zhang, Junyuan Hong, Souvik Kundu, and Zhangyang Wang. 2025c. SEAL: Steerable Reasoning Calibration of Large Language Models for Free. arXiv preprint arXiv:2504.07986 (2025)."},{"key":"e_1_3_2_1_8_1","volume-title":"Zheng Liu, Xu Miao, Yang Lu, Lei Fang, Zhongyuan Wang, and Ji-Rong Wen.","author":"Chen Zhipeng","year":"2025","unstructured":"Zhipeng Chen, Yingqian Min, Beichen Zhang, Jie Chen, Jinhao Jiang, Daixuan Cheng, Wayne Xin Zhao, Zheng Liu, Xu Miao, Yang Lu, Lei Fang, Zhongyuan Wang, and Ji-Rong Wen. 2025a. An Empirical Study on Eliciting and Improving R1-like Reasoning Models. arXiv preprint arXiv:2503.04548 (2025)."},{"key":"e_1_3_2_1_9_1","unstructured":"Alexis Chevalier Jiayi Geng Alexander Wettig Howard Chen Sebastian Mizera... Toni Annala and Danqi Chen. 2024. Language Models as Science Tutors. arXiv:2402.11111 [cs.CL] https:\/\/arxiv.org\/abs\/2402.11111"},{"key":"e_1_3_2_1_10_1","volume-title":"Dolors Ca nabate, and Remigijus Bubnys","author":"Colomer Jordi","year":"2020","unstructured":"Jordi Colomer, Teresa Serra, Dolors Ca nabate, and Remigijus Bubnys. 2020. Reflective learning in higher education: Active methodologies for transformative practices. 3827 pages."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1080\/17439760.2016.1262614"},{"key":"e_1_3_2_1_12_1","unstructured":"Scott Crossley Perpetual Baffour Jules King Lauryn Burleigh Walter Reade and Maggie Demkin. 2024. Learning Agency Lab - Automated Essay Scoring 2.0. https:\/\/kaggle.com\/competitions\/learning-agency-lab-automated-essay-scoring-2. Kaggle Competition."},{"key":"e_1_3_2_1_13_1","unstructured":"CSEDM Workshop. 2019. 2nd Educational Data Mining in Computer Science Education (CSEDM) Workshop. https:\/\/sites.google.com\/asu.edu\/csedm-ws-lak-2019. https:\/\/sites.google.com\/asu.edu\/csedm-ws-lak-2019 In conjunction with LAK 2019 at Arizona State University Tempe AZ USA."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.669"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.bea-1.44"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF01189290"},{"volume-title":"Teacher decision-making in the classroom: A collection of papers","author":"Eggleston John","key":"e_1_3_2_1_17_1","unstructured":"John Eggleston. 2018. Teacher decision-making in the classroom: A collection of papers. Routledge."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1080\/14623943.2021.1968818"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.51219\/JAIMLD\/oluwole-fagbohun\/19"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679760"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2024.100206"},{"volume-title":"Discovery of grounded theory: Strategies for qualitative research","author":"Glaser Barney","key":"e_1_3_2_1_22_1","unstructured":"Barney Glaser and Anselm Strauss. 2017. Discovery of grounded theory: Strategies for qualitative research. Routledge."},{"key":"e_1_3_2_1_23_1","volume-title":"Grounded theory. Strategien qualitativer Forschung","author":"Glaser Barney G","year":"1998","unstructured":"Barney G Glaser and Anselm L Strauss. 1998. Grounded theory. Strategien qualitativer Forschung. Bern: Huber, Vol. 4 (1998)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1080\/0260747920180107"},{"key":"e_1_3_2_1_25_1","volume-title":"Large language models lack essential metacognition for reliable medical reasoning. Nature communications","author":"Griot Maxime","year":"2025","unstructured":"Maxime Griot, Coralie Hemptinne, Jean Vanderdonckt, and Demet Yuksel. 2025. Large language models lack essential metacognition for reliable medical reasoning. Nature communications, Vol. 16, 1 (2025), 642."},{"key":"e_1_3_2_1_26_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1177\/16094069251322346"},{"key":"e_1_3_2_1_28_1","volume-title":"m1: Unleash the Potential of Test-Time Scaling for Medical Reasoning with Large Language Models. arXiv preprint arXiv:2504.00869","author":"Huang Xiaoke","year":"2025","unstructured":"Xiaoke Huang, Juncheng Wu, Hui Liu, Xianfeng Tang, and Yuyin Zhou. 2025. m1: Unleash the Potential of Test-Time Scaling for Medical Reasoning with Large Language Models. arXiv preprint arXiv:2504.00869 (2025)."},{"key":"e_1_3_2_1_29_1","unstructured":"Aaron Jaech Adam Kalai Adam Lerer Adam Richardson Ahmed El-Kishky Aiden Low Alec Helyar Aleksander Madry Alex Beutel Alex Carney et al. 2024. Openai o1 system card. arXiv preprint arXiv:2412.16720 (2024)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10639-023-11834-1"},{"key":"e_1_3_2_1_31_1","volume-title":"Zheng Liu, Dong Yan, Jian Xie, Zhongyuan Wang, and Ji-Rong Wen.","author":"Jiang Jinhao","year":"2024","unstructured":"Jinhao Jiang, Zhipeng Chen, Yingqian Min, Jie Chen, Xiaoxue Cheng, Jiapeng Wang, Yiru Tang, Haoxiang Sun, Jia Deng, Wayne Xin Zhao, Zheng Liu, Dong Yan, Jian Xie, Zhongyuan Wang, and Ji-Rong Wen. 2024. Enhancing LLM Reasoning with Reward-guided Tree Search. arXiv preprint arXiv:2411.11694 (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"CLST: Cold-Start Mitigation in Knowledge Tracing by Aligning a Generative Language Model as a Students' Knowledge Tracer. arXiv preprint arXiv:2406.10296","author":"Jung Heeseok","year":"2024","unstructured":"Heeseok Jung, Jaesang Yoo, Yohaan Yoon, and Yeonju Jang. 2024. CLST: Cold-Start Mitigation in Knowledge Tracing by Aligning a Generative Language Model as a Students' Knowledge Tracer. arXiv preprint arXiv:2406.10296 (2024)."},{"key":"e_1_3_2_1_33_1","unstructured":"Irina Jurenka Markus Kunesch Kevin R. McKee Daniel Gillick Shaojian Zhu... Sara Wiltberger and Lila Ibrahim. 2024a. Towards Responsible Development of Generative AI for Education: An Evaluation-Driven Approach. arXiv:2407.12687 [cs.CY] https:\/\/arxiv.org\/abs\/2407.12687"},{"key":"e_1_3_2_1_34_1","unstructured":"Irina Jurenka Markus Kunesch Kevin R. McKee Daniel Gillick Shaojian Zhu Sara Wiltberger Shubham Milind Phal Katherine Hermann Daniel Kasenberg"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Enkelejda Kasneci Kathrin Se\u00dfler Stefan K\u00fcchemann Maria Bannert Daryna Dementieva Frank Fischer Urs Gasser Georg Groh Stephan G\u00fcnnemann Eyke H\u00fcllermeier et al. 2023. ChatGPT for good? On opportunities and challenges of large language models for education. Learning and individual differences Vol. 103 (2023) 102274.","DOI":"10.1016\/j.lindif.2023.102274"},{"key":"e_1_3_2_1_36_1","volume-title":"Limitations of Large Language Models in Clinical Problem-Solving Arising from Inflexible Reasoning. arXiv preprint arXiv:2502.04381","author":"Kim Jonathan","year":"2025","unstructured":"Jonathan Kim, Anna Podlasek, Kie Shidara, Feng Liu, Ahmed Alaa, and Danilo Bernardo. 2025. Limitations of Large Language Models in Clinical Problem-Solving Arising from Inflexible Reasoning. arXiv preprint arXiv:2502.04381 (2025)."},{"key":"e_1_3_2_1_37_1","volume-title":"Med-r1: Reinforcement learning for generalizable medical reasoning in vision-language models. arXiv preprint arXiv:2503.13939","author":"Lai Yuxiang","year":"2025","unstructured":"Yuxiang Lai, Jike Zhong, Ming Li, Shitian Zhao, and Xiaofeng Yang. 2025. Med-r1: Reinforcement learning for generalizable medical reasoning in vision-language models. arXiv preprint arXiv:2503.13939 (2025)."},{"key":"e_1_3_2_1_38_1","volume-title":"Language model can do knowledge tracing: Simple but effective method to integrate language model and knowledge tracing task. arXiv preprint arXiv:2406.02893","author":"Lee Unggi","year":"2024","unstructured":"Unggi Lee, Jiyeong Bae, Dohee Kim, Sookbun Lee, Jaekwon Park, Taekyung Ahn, Gunho Lee, Damji Stratton, and Hyeoncheol Kim. 2024a. Language model can do knowledge tracing: Simple but effective method to integrate language model and knowledge tracing task. arXiv preprint arXiv:2406.02893 (2024)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2024.100297"},{"key":"e_1_3_2_1_40_1","first-page":"4891","volume-title":"Joint 30th International Conference on Computational Linguistics and 14th International Conference on Language Resources and Evaluation, LREC-COLING","author":"Lee Unggi","year":"2024","unstructured":"Unggi Lee, Sungjun Yoon, Joon Seo Yun, Kyoungsoo Park, Young Hoon Jung, Damji Stratton, and Hyeoncheol Kim. 2024c. Difficulty-Focused Contrastive Learning for Knowledge Tracing with a Large Language Model-Based Difficulty Prediction. In Joint 30th International Conference on Computational Linguistics and 14th International Conference on Language Resources and Evaluation, LREC-COLING 2024. European Language Resources Association (ELRA), 4891-4900."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Zhong-Zhi Li Duzhen Zhang Ming-Liang Zhang Jiaxin Zhang Zengyan Liu Yuxuan Yao Haotian Xu Junhao Zheng Pei-Jie Wang Xiuyi Chen et al. 2025. From system 1 to system 2: A survey of reasoning large language models. arXiv preprint arXiv:2502.17419 (2025).","DOI":"10.1109\/TPAMI.2025.3637037"},{"key":"e_1_3_2_1_42_1","unstructured":"Zhaowei Liu Xin Guo Fangqi Lou Lingfeng Zeng Jinyi Niu Zixuan Wang Jiajie Xu Weige Cai Ziwei Yang Xueqian Zhao et al. 2025. Fin-r1: A large language model for financial reasoning through reinforcement learning. arXiv preprint arXiv:2503.16252 (2025)."},{"key":"e_1_3_2_1_43_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Liu Zitao","year":"2024","unstructured":"Zitao Liu, Qiongqiong Liu, Teng Guo, Jiahao Chen, Shuyan Huang, Xiangyu Zhao, Jiliang Tang, Weiqi Luo, and Jian Weng. 2024. XES3G5M: A Knowledge Tracing Benchmark Dataset with Auxiliary Information. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"volume-title":"Developing reflective practice: Learning about teaching and learning through modelling","author":"Loughran J John","key":"e_1_3_2_1_44_1","unstructured":"J John Loughran. 2002. Developing reflective practice: Learning about teaching and learning through modelling. Routledge."},{"key":"e_1_3_2_1_45_1","volume-title":"MathTutorBench: A Benchmark for Measuring Open-ended Pedagogical Capabilities of LLM Tutors. arXiv preprint arXiv:2502.18940","author":"Macina Jakub","year":"2025","unstructured":"Jakub Macina, Nico Daheim, Ido Hakimi, Manu Kapur, Iryna Gurevych, and Mrinmaya Sachan. 2025. MathTutorBench: A Benchmark for Measuring Open-ended Pedagogical Capabilities of LLM Tutors. arXiv preprint arXiv:2502.18940 (2025)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1080\/02607476.2020.1733398"},{"key":"e_1_3_2_1_47_1","volume-title":"Zheng Liu, Zhongyuan Wang, and Ji-Rong Wen.","author":"Min Yingqian","year":"2024","unstructured":"Yingqian Min, Zhipeng Chen, Jinhao Jiang, Jie Chen, Jia Deng, Yiwen Hu, Yiru Tang, Jiapeng Wang, Xiaoxue Cheng, Huatong Song, Wayne Xin Zhao, Zheng Liu, Zhongyuan Wang, and Ji-Rong Wen. 2024. Imitate, Explore, and Self-Improve: A Reproduction Report on Slow-thinking Reasoning Systems. arXiv preprint arXiv:2412.09413 (2024)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.rmal.2023.100050"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1177\/0049124117729703"},{"key":"e_1_3_2_1_50_1","unstructured":"OpenAI. 2025. Introducing GPT-4.1 in the API. https:\/\/openai.com\/index\/gpt-4-1\/ Accessed: 2025-05-17."},{"key":"e_1_3_2_1_51_1","volume-title":"Chen Chen, Cheng Ouyang, and Daniel Rueckert.","author":"Pan Jiazhen","year":"2025","unstructured":"Jiazhen Pan, Che Liu, Junde Wu, Fenglin Liu, Jiayuan Zhu, Hongwei Bran Li, Chen Chen, Cheng Ouyang, and Daniel Rueckert. 2025. Medvlm-r1: Incentivizing medical reasoning capability of vision-language models (vlms) via reinforcement learning. arXiv preprint arXiv:2502.19634 (2025)."},{"key":"e_1_3_2_1_52_1","volume-title":"Towards the Pedagogical Steering of Large Language Models for Tutoring: A Case Study with Modeling Productive Failure. arXiv preprint arXiv:2410.03781","author":"Puech Romain","year":"2025","unstructured":"Romain Puech, Jakub Macina, Julia Chatain, Mrinmaya Sachan, and Manu Kapur. 2025. Towards the Pedagogical Steering of Large Language Models for Tutoring: A Case Study with Modeling Productive Failure. arXiv preprint arXiv:2410.03781 (2025)."},{"key":"e_1_3_2_1_53_1","first-page":"41","article-title":"The role of ChatGPT in higher education: Benefits, challenges, and future research directions","volume":"6","author":"Rasul Tareq","year":"2023","unstructured":"Tareq Rasul, Sumesh Nair, Diane Kalendra, Mulyadi Robin, Fernando de Oliveira Santini, Wagner Junior Ladeira, Mingwei Sun, Ingrid Day, Raouf Ahmad Rather, and Liz Heathcote. 2023. The role of ChatGPT in higher education: Benefits, challenges, and future research directions. Journal of Applied Learning and Teaching, Vol. 6, 1 (2023), 41-56.","journal-title":"Journal of Applied Learning and Teaching"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cpa.2024.102722"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"crossref","unstructured":"Donald A Sch\u00f6n. 2017. The reflective practitioner: How professionals think in action. (2017).","DOI":"10.4324\/9781315237473"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1177\/0888406416683230"},{"key":"e_1_3_2_1_57_1","first-page":"15","article-title":"The Impact of Reflective Practice on Teacher Candidates","volume":"13","author":"Slade Mary L","year":"2019","unstructured":"Mary L Slade, Tammy J Burnham, Sarah Marie Catalana, and Tammy Waters. 2019. The Impact of Reflective Practice on Teacher Candidates' Learning. International Journal for the Scholarship of Teaching and Learning, Vol. 13, 2 (2019), 15.","journal-title":"Learning. International Journal for the Scholarship of Teaching and Learning"},{"key":"e_1_3_2_1_58_1","volume-title":"Baraniuk","author":"Sonkar Shashank","year":"2024","unstructured":"Shashank Sonkar, Kangqi Ni, Sapana Chaudhary, and Richard G. Baraniuk. 2024. Pedagogical Alignment of Large Language Models. arXiv preprint arXiv:2402.05000 (2024)."},{"key":"e_1_3_2_1_59_1","unstructured":"NovaSky Team. 2025a. Sky-T1: Train your own O1 preview model within $450. https:\/\/novasky-ai.github.io\/posts\/sky-t1. Accessed: 2025-01-09."},{"key":"e_1_3_2_1_60_1","unstructured":"Qwen Team. 2025b. QwQ-32B: Embracing the Power of Reinforcement Learning. https:\/\/qwenlm.github.io\/blog\/qwq-32b\/"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1080\/14623943.2021.1923471"},{"key":"e_1_3_2_1_62_1","volume-title":"Large language models for education: A survey and outlook. arXiv preprint arXiv:2403.18105","author":"Wang Shen","year":"2024","unstructured":"Shen Wang, Tianlong Xu, Hang Li, Chaoli Zhang, Joleen Liang, Jiliang Tang, Philip S Yu, and Qingsong Wen. 2024. Large language models for education: A survey and outlook. arXiv preprint arXiv:2403.18105 (2024)."},{"key":"e_1_3_2_1_63_1","volume-title":"Place: On the Underthinking of o1-Like LLMs. arXiv preprint arXiv:2501.18585","author":"Wang Yue","year":"2025","unstructured":"Yue Wang, Qiuzhi Liu, Jiahao Xu, Tian Liang, Xingyu Chen, Zhiwei He, Linfeng Song, Dian Yu, Juntao Li, Zhuosheng Zhang, et al., 2025. Thoughts Are All Over the Place: On the Underthinking of o1-Like LLMs. arXiv preprint arXiv:2501.18585 (2025)."},{"key":"e_1_3_2_1_64_1","first-page":"1","article-title":"ChatGPT: Challenges, opportunities, and implications for teacher education","volume":"23","author":"Whalen Jeromie","year":"2023","unstructured":"Jeromie Whalen, Chrystalla Mouza, et al., 2023. ChatGPT: Challenges, opportunities, and implications for teacher education. Contemporary Issues in Technology and Teacher Education, Vol. 23, 1 (2023), 1-23.","journal-title":"Contemporary Issues in Technology and Teacher Education"},{"key":"e_1_3_2_1_65_1","unstructured":"Fengli Xu Qianyue Hao Zefang Zong Jingwei Wang Yunke Zhang Jingyi Wang Xiaochong Lan Jiahui Gong Tianjian Ouyang Fanjin Meng et al. 2025. Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models. arXiv preprint arXiv:2501.09686 (2025)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1111\/bjet.13370"},{"key":"e_1_3_2_1_67_1","volume-title":"arXiv preprint arXiv:2412.15115","author":"Yang An","year":"2024","unstructured":"An Yang, Baosong Yang, Beichen Zhang, Binyuan Hui, Bo Zheng, Bowen Yu, Chengyuan Li, Dayiheng Liu, Fei Huang, Haoran Wei, Huan Lin, Jian Yang, Jianhong Tu, Jianwei Zhang, Jianxin Yang, Jiaxi Yang, Jingren Zhou, Junyang Lin, Kai Dang, Keming Lu, Keqin Bao, Kexin Yang, Le Yu, Mei Li, Mingfeng Xue, Pei Zhang, Qin Zhu, Rui Men, Runji Lin, Tianhao Li, Tianyi Tang, Tingyu Xia, Xingzhang Ren, Xuancheng Ren, Yang Fan, Yang Su, Yichang Zhang, Yu Wan, Yuqiong Liu, Zeyu Cui, Zhenru Zhang, and Zihan Qiu. 2024. Qwen2.5 Technical Report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_68_1","volume-title":"Evaluating test-time scaling llms for legal reasoning: Openai o1, deepseek-r1, and beyond. arXiv preprint arXiv:2503.16040","author":"Yu Yaoyao","year":"2025","unstructured":"Yaoyao Yu, Leilei Gan, Yinghao Hu, Bin Wei, Kun Kuang, and Fei Wu. 2025. Evaluating test-time scaling llms for legal reasoning: Openai o1, deepseek-r1, and beyond. arXiv preprint arXiv:2503.16040 (2025)."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-64302-6_13"},{"key":"e_1_3_2_1_70_1","unstructured":"Qian-Wen Zhang Haochen Wang Fang Li Siyu An Lingfeng Qiao Liangcai Gao Di Yin and Xing Sun. 2024. CJEval: A Benchmark for Assessing Large Language Models Using Chinese Junior High School Exam Data. (2024). arXiv:2409.16202 [cs.AI]"},{"key":"e_1_3_2_1_71_1","volume-title":"Med-RLVR: Emerging Medical Reasoning from a 3B base model via reinforcement Learning. arXiv preprint arXiv:2502.19655","author":"Zhang Sheng","year":"2025","unstructured":"Sheng Zhang, Qianchu Liu, Guanghui Qin, Tristan Naumann, and Hoifung Poon. 2025. Med-RLVR: Emerging Medical Reasoning from a 3B base model via reinforcement Learning. arXiv preprint arXiv:2502.19655 (2025)."}],"event":{"name":"CIKM '25: The 34th ACM International Conference on Information and Knowledge Management","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"],"location":"Seoul Republic of Korea","acronym":"CIKM '25"},"container-title":["Proceedings of the 34th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746252.3761133","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T01:49:52Z","timestamp":1765504192000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746252.3761133"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,10]]},"references-count":71,"alternative-id":["10.1145\/3746252.3761133","10.1145\/3746252"],"URL":"https:\/\/doi.org\/10.1145\/3746252.3761133","relation":{},"subject":[],"published":{"date-parts":[[2025,11,10]]},"assertion":[{"value":"2025-11-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}