{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:03:45Z","timestamp":1784138625300,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808628","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"3418-3425","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["<scp>CommonWhy:<\/scp>\n                    A Dataset for Evaluating Entity-Based Causal Commonsense Reasoning in Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-8058-2088","authenticated-orcid":false,"given":"Armin","family":"Toroghi","sequence":"first","affiliation":[{"name":"University of Toronto, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2201-5604","authenticated-orcid":false,"given":"Faeze","family":"Moradi Kalarde","sequence":"additional","affiliation":[{"name":"University of Toronto, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7984-8394","authenticated-orcid":false,"given":"Scott","family":"Sanner","sequence":"additional","affiliation":[{"name":"University of Toronto, Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72."},{"key":"e_1_3_2_1_2_1","volume-title":"CommAI: Evaluating the first steps towards a useful general AI. arXiv preprint arXiv:1701.08954","author":"Baroni Marco","year":"2017","unstructured":"Marco Baroni, Armand Joulin, Allan Jabri, German Kruszewski, Angeliki Lazaridou, Klemen Simonic, and Tomas Mikolov. 2017. CommAI: Evaluating the first steps towards a useful general AI. arXiv preprint arXiv:1701.08954 (2017)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1160"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1160"},{"key":"e_1_3_2_1_5_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Berglund Lukas","year":"2024","unstructured":"Lukas Berglund, Meg Tong, Maximilian Kaufmann, Mikita Balesni, Asa Cooper Stickland, Tomasz Korbak, and Owain Evans. 2024. The Reversal Curse: LLMs trained on ''A is B'' fail to learn ''B is A''. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1376616.1376746"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.63317\/4a6dosi45kib"},{"key":"e_1_3_2_1_9_1","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:2507.06261 (2025)."},{"key":"e_1_3_2_1_10_1","volume-title":"A Balanced Neuro-Symbolic Approach for Commonsense Abductive Logic. arXiv preprint arXiv:2601.18595","author":"Cotnareanu Joseph","year":"2026","unstructured":"Joseph Cotnareanu, Didier Chetelat, Yingxue Zhang, and Mark Coates. 2026. A Balanced Neuro-Symbolic Approach for Commonsense Abductive Logic. arXiv preprint arXiv:2601.18595 (2026)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2701413"},{"key":"e_1_3_2_1_12_1","unstructured":"DeepSeek-AI. 2025. DeepSeek-V3 Technical Report. arXiv preprint (2025). https:\/\/arxiv.org\/abs\/2512.02556"},{"key":"e_1_3_2_1_13_1","volume-title":"NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle: Benchmarks, Emergent Abilities, and Scaling.","author":"Deveci Ethem","year":"2025","unstructured":"?brahim Ethem Deveci and Duygu Ataman. 2025. The Ouroboros of Benchmarking: Reasoning Evaluation in an Era of Saturation. In NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle: Benchmarks, Emergent Abilities, and Scaling."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1162\/TACL_A_00370"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449992"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29770"},{"key":"e_1_3_2_1_17_1","volume-title":"Cr-lt-kgqa: A knowledge graph question answering dataset requiring commonsense reasoning and longtail knowledge. arXiv preprint arXiv:2403.01395","author":"Guo Willis","year":"2024","unstructured":"Willis Guo, Armin Toroghi, and Scott Sanner. 2024. Cr-lt-kgqa: A knowledge graph question answering dataset requiring commonsense reasoning and longtail knowledge. arXiv preprint arXiv:2403.01395 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1162\/TACL.a.47"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.153"},{"key":"e_1_3_2_1_20_1","volume-title":"Wikiwhy: Answering and explaining cause-and-effect questions. arXiv preprint arXiv:2210.12152","author":"Ho Matthew","year":"2022","unstructured":"Matthew Ho, Aditya Sharma, Justin Chang, Michael Saxon, Sharon Levy, Yujie Lu, and William Yang Wang. 2022. Wikiwhy: Answering and explaining cause-and-effect questions. arXiv preprint arXiv:2210.12152 (2022)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447772"},{"key":"e_1_3_2_1_22_1","volume-title":"CRoW: Benchmarking commonsense reasoning in real-world tasks. arXiv preprint arXiv:2310.15239","author":"Ismayilzada Mete","year":"2023","unstructured":"Mete Ismayilzada, Debjit Paul, Syrielle Montariol, Mor Geva, and Antoine Bosselut. 2023. CRoW: Benchmarking commonsense reasoning in real-world tasks. arXiv preprint arXiv:2310.15239 (2023)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.63317\/2ib4rexe35ca"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/511446.511524"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.361"},{"key":"e_1_3_2_1_26_1","volume-title":"S\u00f6ren Auer, et al.","author":"Lehmann Jens","year":"2015","unstructured":"Jens Lehmann, Robert Isele, Max Jakob, Anja Jentzsch, Dimitris Kontokostas, Pablo N Mendes, Sebastian Hellmann, Mohamed Morsey, Patrick Van Kleef, S\u00f6ren Auer, et al. 2015. DBpedia-a large-scale, multilingual knowledge base extracted from wikipedia. Semantic web 6, 2 (2015), 167-195."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 6966-6980","author":"Li Tianle","year":"2023","unstructured":"Tianle Li, Xueguang Ma, Alex Zhuang, Yu Gu, Yu Su, andWenhu Chen. 2023. Fewshot In-context Learning on Knowledge Base Question Answering. In Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). 6966-6980."},{"key":"e_1_3_2_1_28_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval. 3090-3098","author":"Linjordet Trond","year":"2022","unstructured":"Trond Linjordet and Krisztian Balog. 2022. Would you ask it that way? measuring and improving question naturalness for knowledge graph question answering. In Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval. 3090-3098."},{"key":"e_1_3_2_1_30_1","unstructured":"Aixin Liu Aoxue Mei Bangcai Lin Bing Xue Bingxuan Wang Bingzheng Xu Bochao Wu Bowei Zhang Chaofan Lin Chen Dong et al. 2025. Deepseekv3. 2: Pushing the frontier of open large language models. arXiv preprint arXiv:2512.02556 (2025)."},{"key":"e_1_3_2_1_31_1","volume-title":"ConceptNet\u2014a practical commonsense reasoning tool-kit. BT technology journal 22, 4","author":"Liu Hugo","year":"2004","unstructured":"Hugo Liu and Push Singh. 2004. ConceptNet\u2014a practical commonsense reasoning tool-kit. BT technology journal 22, 4 (2004), 211-226."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.dlg4nlp-1.6"},{"key":"e_1_3_2_1_33_1","unstructured":"Meta AI. 2024. LLaMA 3.3 70B Instruct. https:\/\/huggingface.co\/meta-llama\/Llama-3.3-70B-Instruct"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.741"},{"key":"e_1_3_2_1_35_1","volume-title":"The role of logic in knowledge representation and commonsense reasoning. SRI International","author":"Moore Robert C","unstructured":"Robert C Moore. 1982. The role of logic in knowledge representation and commonsense reasoning. SRI International. Artificial Intelligence Center."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1, NeurIPS Datasets and Benchmarks 2021","author":"Onoe Yasumasa","year":"2021","unstructured":"Yasumasa Onoe, Michael J. Q. Zhang, Eunsol Choi, and Greg Durrett. 2021. CREAK: A Dataset for Commonsense Reasoning over Entity Knowledge. In Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1, NeurIPS Datasets and Benchmarks 2021, December 2021, virtual, Joaquin Vanschoren and Sai-Kit Yeung (Eds.). https:\/\/datasets-benchmarks-proceedings.neurips.cc\/paper\/2021\/hash\/5737c6ec2e0716f3d8a7a5c4e0de0d9a-Abstract-round2.html"},{"key":"e_1_3_2_1_37_1","unstructured":"OpenAI. 2024. GPT-4o Technical Report. (2024). https:\/\/cdn.openai.com\/gpt-4osystem-card.pdf"},{"key":"e_1_3_2_1_38_1","unstructured":"OpenAI. 2025. GPT-5.1. https:\/\/openai.com\/index\/gpt-5-1\/"},{"key":"e_1_3_2_1_39_1","unstructured":"OpenAI. 2025. OpenAI o3. https:\/\/platform.openai.com\/docs\/models\/o3"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.185"},{"key":"e_1_3_2_1_41_1","volume-title":"Personal health knowledge graphs for patients. arXiv preprint arXiv:2004.00071","author":"Rastogi Nidhi","year":"2020","unstructured":"Nidhi Rastogi and Mohammed J Zaki. 2020. Personal health knowledge graphs for patients. arXiv preprint arXiv:2004.00071 (2020)."},{"key":"e_1_3_2_1_42_1","volume-title":"A comprehensive review of recommender systems: Transitioning from theory to practice. arXiv preprint arXiv:2407.13699","author":"Raza Shaina","year":"2024","unstructured":"Shaina Raza, Mizanur Rahman, Safiullah Kamawal, Armin Toroghi, Ananya Raval, Farshad Navah, and Amirmohammad Kazemeini. 2024. A comprehensive review of recommender systems: Transitioning from theory to practice. arXiv preprint arXiv:2407.13699 (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"Cosmin Adrian Bejan, and Andrew S Gordon","author":"Roemmele Melissa","year":"2011","unstructured":"Melissa Roemmele, Cosmin Adrian Bejan, and Andrew S Gordon. 2011. Choice of Plausible Alternatives: An Evaluation of Commonsense Causal Reasoning.. In AAAI spring symposium: logical formalizations of commonsense reasoning. 90-95."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013027"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1454"},{"key":"e_1_3_2_1_46_1","volume-title":"W3C","author":"Seaborne Andy","year":"2008","unstructured":"Andy Seaborne and Eric Prud'hommeaux. 2008. SPARQL query language for RDF. W3C Recommendation, W3C (2008)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1609\/AAAI.V31I1.11164"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/N19-1421"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3592012"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591954"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.379"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.378"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730291"},{"key":"e_1_3_2_1_54_1","volume-title":"The Thirteenth International Conference on Learning Representations.","author":"Toroghi Armin","year":"2025","unstructured":"Armin Toroghi, Ali Pesaranghader, Tanmana Sadhu, and Scott Sanner. 2025. Llmbased typed hyperresolution for commonsense reasoning with knowledge bases. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i18.30040"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-68204-4_22"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/2629489"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3018"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.ijcnlp-long.148"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/P16-2033"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers). 201-206","author":"Richardson Matthew","year":"2016","unstructured":"Wen-tau Yih, Matthew Richardson, Christopher Meek, Ming-Wei Chang, and Jina Suh. 2016. The value of semantic parse labeling for knowledge base question answering. In Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers). 201-206."},{"key":"e_1_3_2_1_62_1","volume-title":"BERTScore: Evaluating Text Generation with BERT. In International Conference on Learning Representations.","author":"Zhang Tianyi","year":"2020","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian Q Weinberger, and Yoav Artzi. 2020. BERTScore: Evaluating Text Generation with BERT. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12057"},{"key":"e_1_3_2_1_64_1","volume-title":"Wee Sun Lee, and David Hsu","author":"Zhao Zirui","year":"2023","unstructured":"Zirui Zhao, Wee Sun Lee, and David Hsu. 2023. Large language models as commonsense knowledge for large-scale task planning. Advances in neural information processing systems 36 (2023), 31967-31987."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3132847.3132977"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:07:15Z","timestamp":1784135235000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808628"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":65,"alternative-id":["10.1145\/3805712.3808628","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808628","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}