{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:10:19Z","timestamp":1784139019481,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808551","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"2942-2951","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A Reproducibility Study of Metacognitive Retrieval-Augmented Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-9635-0683","authenticated-orcid":false,"given":"Gabriel","family":"Iturra Bocaz","sequence":"first","affiliation":[{"name":"University of Stavanger, Stavanger, Rogaland, Norway"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6328-7131","authenticated-orcid":false,"given":"Petra","family":"Galu\u0161\u010d\u00e1kov\u00e1","sequence":"additional","affiliation":[{"name":"University of Stavanger, Stavanger, Rogaland, Norway"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Self-rag: Learning to retrieve, generate, and critique through self-reflection.","author":"Asai Akari","year":"2024","unstructured":"Akari Asai, Zeqiu Wu, Yizhong Wang, Avirup Sil, and Hannaneh Hajishirzi. 2024. Self-rag: Learning to retrieve, generate, and critique through self-reflection. (2024)."},{"key":"e_1_3_2_1_2_1","volume-title":"Evaluating chatgpt as a question answering system: A comprehensive analysis and comparison with existing models. arXiv preprint arXiv:2312.07592","author":"Bahak Hossein","year":"2023","unstructured":"Hossein Bahak, Farzaneh Taheri, Zahra Zojaji, and Arefeh Kazemi. 2023. Evaluating chatgpt as a question answering system: A comprehensive analysis and comparison with existing models. arXiv preprint arXiv:2312.07592 (2023)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Jianlv Chen Shitao Xiao Peitian Zhang Kun Luo Defu Lian and Zheng Liu. 2024. BGE M3-Embedding: Multi-Lingual Multi-Functionality Multi-Granularity Text Embeddings Through Self-Knowledge Distillation. arXiv:2402.03216 [cs.CL]","DOI":"10.18653\/v1\/2024.findings-acl.137"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1571941.1572114"},{"key":"e_1_3_2_1_5_1","volume-title":"Generate. In Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 2701-2715","author":"Glass Michael","year":"2022","unstructured":"Michael Glass, Gaetano Rossiello, Md Faisal Mahbub Chowdhury, Ankita Naik, Pengshan Cai, and Alfio Gliozzo. 2022. Re2G: Retrieve, Rerank, Generate. In Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 2701-2715."},{"key":"e_1_3_2_1_6_1","volume-title":"International association for development of the information society","author":"Gotoh Yasushi","year":"2016","unstructured":"Yasushi Gotoh. 2016. Development of Critical Thinking with Metacognitive Regulation. International association for development of the information society (2016)."},{"key":"e_1_3_2_1_7_1","volume-title":"Efficient Nearest Neighbor Language Models. In Conference on Empirical Methods in Natural Language Processing.","author":"He Junxian","year":"2021","unstructured":"Junxian He, Graham Neubig, and Taylor Berg-Kirkpatrick. 2021. Efficient Nearest Neighbor Language Models. In Conference on Empirical Methods in Natural Language Processing."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.580"},{"key":"e_1_3_2_1_9_1","first-page":"874","article-title":"Leveraging Passage Retrieval with Generative Models for Open Domain Question Answering. In EACL 2021-16th Conference of the European Chapter of the Association for Computational Linguistics","author":"Izacard Gautier","year":"2021","unstructured":"Gautier Izacard and Edouard Grave. 2021. Leveraging Passage Retrieval with Generative Models for Open Domain Question Answering. In EACL 2021-16th Conference of the European Chapter of the Association for Computational Linguistics. Association for Computational Linguistics, 874-880.","journal-title":"Association for Computational Linguistics"},{"key":"e_1_3_2_1_10_1","volume-title":"Sung Ju Hwang, and Jong C Park","author":"Jeong Soyeong","year":"2024","unstructured":"Soyeong Jeong, Jinheon Baek, Sukmin Cho, Sung Ju Hwang, and Jong C Park. 2024. Adaptive-rag: Learning to adapt retrieval-augmented large language models through question complexity. arXiv preprint arXiv:2403.14403 (2024)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.495"},{"key":"e_1_3_2_1_12_1","volume-title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning. arXiv preprint arXiv:2503.09516","author":"Jin Bowen","year":"2025","unstructured":"Bowen Jin, Hansi Zeng, Zhenrui Yue, Jinsung Yoon, Sercan Arik, Dong Wang, Hamed Zamani, and Jiawei Han. 2025. Search-r1: Training llms to reason and leverage search engines with reinforcement learning. arXiv preprint arXiv:2503.09516 (2025)."},{"key":"e_1_3_2_1_13_1","volume-title":"Triviaqa: A large scale distantly supervised challenge dataset for reading comprehension. arXiv preprint arXiv:1705.03551","author":"Joshi Mandar","year":"2017","unstructured":"Mandar Joshi, Eunsol Choi, Daniel S Weld, and Luke Zettlemoyer. 2017. Triviaqa: A large scale distantly supervised challenge dataset for reading comprehension. arXiv preprint arXiv:1705.03551 (2017)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-45442-5_4"},{"key":"e_1_3_2_1_15_1","volume-title":"Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih.","author":"Karpukhin Vladimir","year":"2020","unstructured":"Vladimir Karpukhin, Barlas Oguz, Sewon Min, Patrick SH Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih. 2020. Dense Passage Retrieval for Open-Domain Question Answering. In EMNLP (1). 6769-6781."},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Learning Representations.","author":"Khandelwal Urvashi","unstructured":"Urvashi Khandelwal, Omer Levy, Dan Jurafsky, Luke Zettlemoyer, and Mike Lewis. [n.d.]. Generalization through Memorization: Nearest Neighbor Language Models. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_17_1","volume-title":"Decomposed Prompting: A Modular Approach for Solving Complex Tasks. In The Eleventh International Conference on Learning Representations.","author":"Khot Tushar","unstructured":"Tushar Khot, Harsh Trivedi, Matthew Finlayson, Yao Fu, Kyle Richardson, Peter Clark, and Ashish Sabharwal. [n.d.]. Decomposed Prompting: A Modular Approach for Solving Complex Tasks. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_18_1","volume-title":"Metacognition: A literature review.","author":"Lai Emily R","year":"2011","unstructured":"Emily R Lai. 2011. Metacognition: A literature review. (2011)."},{"key":"e_1_3_2_1_19_1","volume-title":"Haidar Khan, Israt Jahan, Amran Bhuiyan, Chee Wei Tan, Md Rizwan Parvez, et al.","author":"Rahman Laskar Md Tahmid","year":"2024","unstructured":"Md Tahmid Rahman Laskar, Sawsan Alqahtani, M Saiful Bari, Mizanur Rahman, Mohammad Abdullah Matin Khan, Haidar Khan, Israt Jahan, Amran Bhuiyan, Chee Wei Tan, Md Rizwan Parvez, et al. 2024. A systematic survey and critical review on evaluating large language models: Challenges, limitations, and recommendations. arXiv preprint arXiv:2407.04069 (2024)."},{"key":"e_1_3_2_1_20_1","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen-tau Yih Tim Rockt\u00e4schel et al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in neural information processing systems 33 (2020) 9459-9474."},{"key":"e_1_3_2_1_21_1","volume-title":"SMART-RAG: Selection using Determinantal Matrices for Augmented Retrieval. arXiv preprint arXiv:2409.13992","author":"Li Jiatao","year":"2024","unstructured":"Jiatao Li, Xinyu Hu, and Xiaojun Wan. 2024. SMART-RAG: Selection using Determinantal Matrices for Augmented Retrieval. arXiv preprint arXiv:2409.13992 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"When hindsight is not 20\/20: Testing limits on reflective thinking in large language models. arXiv preprint arXiv:2404.09129","author":"Li Yanhong","year":"2024","unstructured":"Yanhong Li, Chenghao Yang, and Allyson Ettinger. 2024. When hindsight is not 20\/20: Testing limits on reflective thinking in large language models. arXiv preprint arXiv:2404.09129 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Towards general text embeddings with multi-stage contrastive learning. arXiv preprint arXiv:2308.03281","author":"Li Zehan","year":"2023","unstructured":"Zehan Li, Xin Zhang, Yanzhao Zhang, Dingkun Long, Pengjun Xie, and Meishan Zhang. 2023. Towards general text embeddings with multi-stage contrastive learning. arXiv preprint arXiv:2308.03281 (2023)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Vaibhav Mavi Anubhav Jangra Adam Jatowt et al. 2024. Multi-hop question answering. Foundations and Trends\u00ae in Information Retrieval 17 5 (2024) 457-586.","DOI":"10.1561\/1500000102"},{"key":"e_1_3_2_1_25_1","volume-title":"Benchmarking, fine-tuning and deploying Rerankers for RAG. arXiv preprint arXiv:2409.07691","author":"Moreira Gabriel","year":"2024","unstructured":"Gabriel de Souza P Moreira, Ronay Ak, Benedikt Schifferer, Mengyao Xu, Radek Osmulski, and Even Oldridge. 2024. Enhancing Q&A Text Retrieval with Ranking Models: Benchmarking, fine-tuning and deploying Rerankers for RAG. arXiv preprint arXiv:2409.07691 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"Metamemory: A theoretical framework and some new findings. The Psychology of Learning and Motivation.","author":"Nelson TO","year":"1990","unstructured":"TO Nelson and L Narens. 1990. Metamemory: A theoretical framework and some new findings. The Psychology of Learning and Motivation. Vol. 26."},{"key":"e_1_3_2_1_27_1","unstructured":"Tri Nguyen Mir Rosenberg Xia Song Jianfeng Gao Saurabh Tiwary Rangan Majumder and Li Deng. 2016. Ms marco: A human-generated machine reading comprehension dataset. (2016)."},{"key":"e_1_3_2_1_28_1","volume-title":"RankZephyr: Effective and Robust Zero-Shot Listwise Reranking is a Breeze! arXiv preprint arXiv:2312.02724","author":"Pradeep Ronak","year":"2023","unstructured":"Ronak Pradeep, Sahel Sharifymoghaddam, and Jimmy Lin. 2023. RankZephyr: Effective and Robust Zero-Shot Listwise Reranking is a Breeze! arXiv preprint arXiv:2312.02724 (2023)."},{"key":"e_1_3_2_1_29_1","volume-title":"Measuring and narrowing the compositionality gap in language models. arXiv preprint arXiv:2210.03350","author":"Press Ofir","year":"2022","unstructured":"Ofir Press, Muru Zhang, Sewon Min, Ludwig Schmidt, Noah A Smith, and Mike Lewis. 2022. Measuring and narrowing the compositionality gap in language models. arXiv preprint arXiv:2210.03350 (2022)."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the Third Workshop on Natural Language Generation, Evaluation, and Metrics (GEM). 138-154","author":"Rabinovich Ella","year":"2023","unstructured":"Ella Rabinovich, Samuel Ackerman, Orna Raz, Eitan Farchi, and Ateret Anaby Tavor. 2023. Predicting question-answering performance of large language models through semantic consistency. In Proceedings of the Third Workshop on Natural Language Generation, Evaluation, and Metrics (GEM). 138-154."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-88714-7_29"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Stephen Robertson Hugo Zaragoza et al. 2009. The probabilistic relevance framework: BM25 and beyond. Foundations and Trends\u00ae in Information Retrieval 3 4 (2009) 333-389.","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_1_33_1","volume-title":"Metacognitive theories. Educational psychology review 7, 4","author":"Schraw Gregory","year":"1995","unstructured":"Gregory Schraw and David Moshman. 1995. Metacognitive theories. Educational psychology review 7, 4 (1995), 351-371."},{"key":"e_1_3_2_1_34_1","unstructured":"Sander Schulhoff Michael Ilie Nishant Balepur Konstantine Kahadze Amanda Liu Chenglei Si Yinheng Li Aayush Gupta HyoJung Han Sevien Schulhoff et al. 2024. The prompt report: a systematic survey of prompt engineering techniques. arXiv preprint arXiv:2406.06608 (2024)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730331"},{"key":"e_1_3_2_1_36_1","volume-title":"Reflexion: Language agents with verbal reinforcement learning","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Federico Cassano, Beck Labash, Ashwin Gopinath, Karthik Narasimhan, and Shunyu Yao. 2023. Reflexion: Language agents with verbal reinforcement learning, 2023. URL https:\/\/arxiv.org\/abs\/2303.11366 1 (2023)."},{"key":"e_1_3_2_1_37_1","volume-title":"Is ChatGPT good at search? investigating large language models as re-ranking agents. arXiv preprint arXiv:2304.09542","author":"Sun Weiwei","year":"2023","unstructured":"Weiwei Sun, Lingyong Yan, Xinyu Ma, Shuaiqiang Wang, Pengjie Ren, Zhumin Chen, Dawei Yin, and Zhaochun Ren. 2023. Is ChatGPT good at search? investigating large language models as re-ranking agents. arXiv preprint arXiv:2304.09542 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"Multihop-rag: Benchmarking retrievalaugmented generation for multi-hop queries. arXiv preprint arXiv:2401.15391","author":"Tang Yixuan","year":"2024","unstructured":"Yixuan Tang and Yi Yang. 2024. Multihop-rag: Benchmarking retrievalaugmented generation for multi-hop queries. arXiv preprint arXiv:2401.15391 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Interleaving retrieval with chain-of-thought reasoning for knowledgeintensive multi-step questions. arXiv preprint arXiv:2212.10509","author":"Trivedi Harsh","year":"2022","unstructured":"Harsh Trivedi, Niranjan Balasubramanian, Tushar Khot, and Ashish Sabharwal. 2022. Interleaving retrieval with chain-of-thought reasoning for knowledgeintensive multi-step questions. arXiv preprint arXiv:2212.10509 (2022)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.557"},{"key":"e_1_3_2_1_41_1","volume-title":"Text embeddings by weakly-supervised contrastive pre-training. arXiv preprint arXiv:2212.03533","author":"Wang Liang","year":"2022","unstructured":"Liang Wang, Nan Yang, Xiaolong Huang, Binxing Jiao, Linjun Yang, Daxin Jiang, Rangan Majumder, and Furu Wei. 2022. Text embeddings by weakly-supervised contrastive pre-training. arXiv preprint arXiv:2212.03533 (2022)."},{"key":"e_1_3_2_1_42_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Schuurmans Dale","year":"2022","unstructured":"XuezhiWang, JasonWei, Dale Schuurmans, et al. 2022. Self-consistency improves chain of thought reasoning in language models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_43_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730018"},{"key":"e_1_3_2_1_45_1","volume-title":"HotpotQA: A dataset for diverse, explainable multi-hop question answering. arXiv preprint arXiv:1809.09600","author":"Yang Zhilin","year":"2018","unstructured":"Zhilin Yang, Peng Qi, Saizheng Zhang, Yoshua Bengio, WilliamWCohen, Ruslan Salakhutdinov, and Christopher D Manning. 2018. HotpotQA: A dataset for diverse, explainable multi-hop question answering. arXiv preprint arXiv:1809.09600 (2018)."},{"key":"e_1_3_2_1_46_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao. 2023. React: Synergizing reasoning and acting in language models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_47_1","unstructured":"Tianyu Yu et al. 2025. Unleashing the Power of Context Repetition for Robust Reasoning in LLMs. arXiv preprint arXiv:2503.06789 (2025)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3850"},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: Industry Track. 1393-1412","author":"Zhang Xin","year":"2024","unstructured":"Xin Zhang, Yanzhao Zhang, Dingkun Long, Wen Xie, Ziqi Dai, Jialong Tang, Huan Lin, Baosong Yang, Pengjun Xie, Fei Huang, et al. 2024. mGTE: Generalized Long-Context Text Representation and Reranking Models for Multilingual Text Retrieval. In Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: Industry Track. 1393-1412."},{"key":"e_1_3_2_1_50_1","first-page":"5657","volume-title":"Training Language Models with Memory Augmentation. In 2022 Conference on Empirical Methods in Natural Language Processing, EMNLP","author":"Zhong Zexuan","year":"2022","unstructured":"Zexuan Zhong, Tao Lei, and Danqi Chen. 2022. Training Language Models with Memory Augmentation. In 2022 Conference on Empirical Methods in Natural Language Processing, EMNLP 2022. Association for Computational Linguistics (ACL), 5657-5673."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645481"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:30:29Z","timestamp":1784136629000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808551"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":51,"alternative-id":["10.1145\/3805712.3808551","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808551","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}