{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T19:25:26Z","timestamp":1743103526692,"version":"3.40.3"},"publisher-location":"Cham","reference-count":26,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031711695"},{"type":"electronic","value":"9783031711701"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-71170-1_18","type":"book-chapter","created":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T12:02:14Z","timestamp":1725883334000},"page":"207-221","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reasoning in\u00a0Transformers \u2013 Mitigating Spurious Correlations and\u00a0Reasoning Shortcuts"],"prefix":"10.1007","author":[{"given":"Daniel","family":"Enstr\u00f6m","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Viktor","family":"Kjellberg","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1097-8278","authenticated-orcid":false,"given":"Moa","family":"Johansson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,10]]},"reference":[{"key":"18_CR1","doi-asserted-by":"crossref","unstructured":"Britz, D., Goldie, A., Luong, M.T., Le, Q.V.: Massive exploration of neural machine translation architectures. arXiv preprint arXiv:1703.03906 (2017)","DOI":"10.18653\/v1\/D17-1151"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Clark, P., Tafjord, O., Richardson, K.: Transformers as soft reasoners over language. In: Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence. IJCAI\u201920 (2021)","DOI":"10.24963\/ijcai.2020\/537"},{"key":"18_CR3","unstructured":"Creswell, A., Shanahan, M., Higgins, I.: Selection-inference: exploiting large language models for interpretable logical reasoning. In: The Eleventh International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=3Pf3Wg6o-A4"},{"key":"18_CR4","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), Minneapolis, Minnesota, pp. 4171\u20134186. Association for Computational Linguistics (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1423, https:\/\/aclanthology.org\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"18_CR5","doi-asserted-by":"publisher","unstructured":"Elazar, Y., et al.: Measuring and improving consistency in pretrained language models. Trans. Assoc. Comput. Linguistics 9, 1012\u20131031 (2021). https:\/\/doi.org\/10.1162\/tacl_a_00410, https:\/\/aclanthology.org\/2021.tacl-1.60","DOI":"10.1162\/tacl_a_00410"},{"key":"18_CR6","unstructured":"Enstr\u00f6m, D., Kjellberg, V.: Approximating reasoning with transformer language models (2023). https:\/\/gupea.ub.gu.se\/handle\/2077\/78873, MSc thesis Gothenburg University"},{"key":"18_CR7","unstructured":"Geng, S., Josifosky, M., Peyrard, M., West, R.: Flexible grammar-based constrained decoding for language models (2023)"},{"key":"18_CR8","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. In: Oh, A.H., Agarwal, A., Belgrave, D., Cho, K. (eds.) Advances in Neural Information Processing Systems (2022), https:\/\/openreview.net\/forum?id=e2TBb5y0yFf"},{"key":"18_CR9","doi-asserted-by":"publisher","unstructured":"Lewis, M., et al.: BART: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 7871\u20137880. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.703, https:\/\/aclanthology.org\/2020.acl-main.703","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"18_CR10","unstructured":"Marconato, E., Bontempo, G., Ficarra, E., Calderara, S., Passerini, A., Teso, S.: Neuro-symbolic continual learning: knowledge, reasoning shortcuts and concept rehearsal. In: Proceedings of the 40th International Conference on Machine Learning. ICML\u201923, JMLR.org (2023)"},{"key":"18_CR11","unstructured":"Marconato, E., Teso, S., Vergari, A., Passerini, A.: Not all neuro-symbolic concepts are created equal: analysis and mitigation of reasoning shortcuts. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. NIPS \u201923, Curran Associates Inc., Red Hook (2023)"},{"key":"18_CR12","unstructured":"Nye, M., et al.: Show your work: scratchpads for intermediate computation with language models (2021)"},{"key":"18_CR13","doi-asserted-by":"publisher","unstructured":"Olausson, T., et al.: LINC: a neurosymbolic approach for logical reasoning by combining language models with first-order logic provers. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, Singapore, pp. 5153\u20135176. Association for Computational Linguistics (2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.313, https:\/\/aclanthology.org\/2023.emnlp-main.313","DOI":"10.18653\/v1\/2023.emnlp-main.313"},{"key":"18_CR14","doi-asserted-by":"publisher","unstructured":"Polu, S., Sutskever, I.: Generative language modeling for automated theorem proving (2020). https:\/\/doi.org\/10.48550\/ARXIV.2009.03393, arXiv:abs\/2009.03393","DOI":"10.48550\/ARXIV.2009.03393"},{"key":"18_CR15","unstructured":"Prystawski, B., Goodman, N.D.: Why think step-by-step? Reasoning emerges from the locality of experience (2023)"},{"key":"18_CR16","unstructured":"PyTorch Development Team: Reproducibility. https:\/\/pytorch.org\/docs\/stable\/notes\/randomness.html (Accessed 2023). Accessed 23 May 2023"},{"key":"18_CR17","doi-asserted-by":"publisher","unstructured":"Rabe, M.N., Lee, D., Bansal, K., Szegedy, C.: Mathematical reasoning via self-supervised skip-tree training. arXiv: Learning (2020). https:\/\/doi.org\/10.48550\/ARXIV.2006.04757, arXiv:abs\/2006.04757","DOI":"10.48550\/ARXIV.2006.04757"},{"issue":"8","key":"18_CR18","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"18_CR19","unstructured":"Saxton, D., Grefenstette, E., Hill, F., Kohli, P.: Analysing mathematical reasoning abilities of neural models. In: International Conference on Learning Representations (2019). https:\/\/openreview.net\/forum?id=H1gR5iR5FX"},{"key":"18_CR20","unstructured":"Susskind, Z., Arden, B., John, L.K., Stockton, P.A., John, E.B.: Neuro-symbolic AI: an emerging class of AI workloads and their characterization. arXiv:abs\/2109.06133 (2021)"},{"key":"18_CR21","doi-asserted-by":"publisher","unstructured":"Tafjord, O., Dalvi, B., Clark, P.: ProofWriter: generating implications, proofs, and abductive statements over natural language. In: Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021, pp. 3621\u20133634. Association for Computational Linguistics (2021). https:\/\/doi.org\/10.18653\/v1\/2021.findings-acl.317, https:\/\/aclanthology.org\/2021.findings-acl.317","DOI":"10.18653\/v1\/2021.findings-acl.317"},{"key":"18_CR22","unstructured":"Talmor, A., Tafjord, O., Clark, P., Goldberg, Y., Berant, J.: Leap-of-thought: teaching pre-trained models to systematically reason over implicit knowledge. In: Proceedings of the 34th International Conference on Neural Information Processing Systems. NIPS\u201920, Red Hook, NY, USA. Curran Associates Inc. (2020)"},{"key":"18_CR23","doi-asserted-by":"publisher","unstructured":"Valmeekam, K., Olmo, A., Sreedharan, S., Kambhampati, S.: Large language models still can\u2019t plan (a benchmark for LLMs on planning and reasoning about change). ArXiv (2022). https:\/\/doi.org\/10.48550\/ARXIV.2206.10498, arXiv:arXiv:abs\/2206.10498","DOI":"10.48550\/ARXIV.2206.10498"},{"key":"18_CR24","doi-asserted-by":"publisher","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, pp. 6000\u20136010. NIPS\u201917, Red Hook, NY, USA. Curran Associates Inc. (2017). https:\/\/doi.org\/10.48550\/ARXIV.1706.03762, arXiv:abs\/1706.03762","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"18_CR25","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q.V., Zhou, D.: Chain-of-thought prompting elicits reasoning in large language models. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems. vol.\u00a035, pp. 24824\u201324837. Curran Associates, Inc. (2022), https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/9d5609613524ecf4f15af0f7b31abca4-Paper-Conference.pdf"},{"key":"18_CR26","doi-asserted-by":"publisher","unstructured":"Zhang, H., Li, L.H., Meng, T., Chang, K., den Broeck, G.V.: On the paradox of learning to reason from data. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence, IJCAI 2023, 19th-25th August 2023, Macao, SAR, China, pp. 3365\u20133373. ijcai.org (2023). https:\/\/doi.org\/10.24963\/ijcai.2023\/375","DOI":"10.24963\/ijcai.2023\/375"}],"container-title":["Lecture Notes in Computer Science","Neural-Symbolic Learning and Reasoning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-71170-1_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T12:07:09Z","timestamp":1725883629000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-71170-1_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031711695","9783031711701"],"references-count":26,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-71170-1_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"10 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NeSy","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural-Symbolic Learning and Reasoning","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Barcelona","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nesy2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/sites.google.com\/view\/nesy2023","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}