{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T15:14:49Z","timestamp":1783696489163,"version":"3.55.0"},"reference-count":55,"publisher":"Elsevier BV","issue":"7","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100016172","name":"Luxembourg Institute of Science and Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100016172","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005050","name":"Mitutoyo Association for Science and Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005050","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006433","name":"Barcelona Supercomputing Center - National Supercomputing Center","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100006433","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Processing &amp; Management"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.ipm.2026.104878","type":"journal-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T10:18:01Z","timestamp":1777976281000},"page":"104878","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["Cross-platform evaluation of reasoning capabilities in foundation models"],"prefix":"10.1016","volume":"63","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8334-4719","authenticated-orcid":false,"given":"J.","family":"de Curt\u00f2","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5844-7871","authenticated-orcid":false,"given":"I.","family":"de Zarz\u00e0","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pablo","family":"Garc\u00eda","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2418-2489","authenticated-orcid":false,"given":"Jordi","family":"Cabot","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0038-0539","authenticated-orcid":false,"given":"Juan Carlos","family":"Cano","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5729-3041","authenticated-orcid":false,"given":"Carlos T.","family":"Calafate","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.ipm.2026.104878_b1","series-title":"Phi-3 technical report: A highly capable language model locally on your phone","author":"Abdin","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b2","series-title":"GPT-OSS-120B cross-provider performance benchmark: AIME 2025 evaluation","author":"Artificial Analysis","year":"2025"},{"key":"10.1016\/j.ipm.2026.104878_b3","series-title":"Program synthesis with large language models","author":"Austin","year":"2021"},{"key":"10.1016\/j.ipm.2026.104878_b4","series-title":"Training a helpful and harmless assistant with reinforcement learning from human feedback","author":"Bai","year":"2022"},{"key":"10.1016\/j.ipm.2026.104878_b5","doi-asserted-by":"crossref","unstructured":"Bender, E. M., Gebru, T., McMillan-Major, A. Shmitchell, S. (2021). On the dangers of stochastic parrots: Can language models be too big?. In Proceedings of the 2021 ACM conference on fairness, accountability, and transparency (pp. 610\u2013623).","DOI":"10.1145\/3442188.3445922"},{"key":"10.1016\/j.ipm.2026.104878_b6","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.ipm.2026.104878_b7","series-title":"Mind the language gap: Automated and augmented evaluation of bias in LLMs for high-and low-resource languages","author":"Buscemi","year":"2025"},{"key":"10.1016\/j.ipm.2026.104878_b8","series-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"issue":"240","key":"10.1016\/j.ipm.2026.104878_b9","first-page":"1","article-title":"Palm: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.ipm.2026.104878_b10","unstructured":"Clark, P., Cowhey, I., Etzioni, O., Khot, T., Sabharwal, A., Schoenick, C., & Tafjord, O. (2018). Think you have solved question answering? Try ARC, the AI2 reasoning challenge. vol. 32, In Proceedings of the AAAI conference on artificial intelligence."},{"key":"10.1016\/j.ipm.2026.104878_b11","series-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021"},{"key":"10.1016\/j.ipm.2026.104878_b12","doi-asserted-by":"crossref","unstructured":"de Curt\u00f2, J., & de Zarz\u00e0, I. (2024). Comparative analysis of reasoning capabilities in foundation models. In 2024 2nd International Conference on Foundation and Large Language Models (FLLM) (pp. 141\u2013149). http:\/\/dx.doi.org\/10.1109\/FLLM63129.2024.10852449.","DOI":"10.1109\/FLLM63129.2024.10852449"},{"key":"10.1016\/j.ipm.2026.104878_b13","doi-asserted-by":"crossref","first-page":"214772","DOI":"10.1109\/ACCESS.2025.3646270","article-title":"Metamorphic testing for semantic invariance in large language models","volume":"13","author":"De Curt\u00f2","year":"2025","journal-title":"IEEE Access"},{"key":"10.1016\/j.ipm.2026.104878_b14","series-title":"KES international symposium on agent and multi-agent systems: technologies and applications","first-page":"3","article-title":"LLM multi-agent decision optimization","author":"de Curt\u00f2","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b15","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.-W., Lee, K., & Toutanova, K. (2019). Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers) (pp. 4171\u20134186).","DOI":"10.18653\/v1\/N19-1423"},{"issue":"120","key":"10.1016\/j.ipm.2026.104878_b16","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.ipm.2026.104878_b17","series-title":"First conference on language modeling","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b18","doi-asserted-by":"crossref","first-page":"1019","DOI":"10.1613\/jair.1.16905","article-title":"Improving reproducibility in AI research: Four mechanisms adopted by jair","volume":"81","author":"Gundersen","year":"2024","journal-title":"Journal of Artificial Intelligence Research"},{"key":"10.1016\/j.ipm.2026.104878_b19","doi-asserted-by":"crossref","unstructured":"Gundersen, O. E., & Kjensmo, S. (2018). State of the art: Reproducibility in artificial intelligence. vol. 32, In Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v32i1.11503"},{"key":"10.1016\/j.ipm.2026.104878_b20","series-title":"Proceedings of the neural information processing systems track on datasets and benchmarks","article-title":"Measuring mathematical problem solving with the MATH dataset","volume":"vol. 1","author":"Hendrycks","year":"2021"},{"key":"10.1016\/j.ipm.2026.104878_b21","series-title":"Advances in neural information processing systems","first-page":"30016","article-title":"An empirical analysis of compute-optimal large language model training","volume":"vol. 35","author":"Hoffmann","year":"2022"},{"key":"10.1016\/j.ipm.2026.104878_b22","series-title":"International conference on machine learning","first-page":"4411","article-title":"XTREME: A massively multilingual multi-task benchmark for evaluating cross-lingual generalisation","author":"Hu","year":"2020"},{"issue":"5","key":"10.1016\/j.ipm.2026.104878_b23","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2025.104152","article-title":"Integrative modeling enables ChatGPT to achieve average level of human counselors performance in mental health Q&A","volume":"62","author":"Huang","year":"2025","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.104878_b24","series-title":"The twelfth international conference on learning representations","article-title":"SWE-bench: Can language models resolve real-world github issues?","author":"Jimenez","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b25","series-title":"Thinking, fast and slow","author":"Kahneman","year":"2011"},{"key":"10.1016\/j.ipm.2026.104878_b26","series-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"10.1016\/j.ipm.2026.104878_b27","series-title":"Findings of the association for computational linguistics: ACL 2025","first-page":"25853","article-title":"Memorization vs. Reasoning: Updating LLMs with new knowledge","author":"Li","year":"2025"},{"key":"10.1016\/j.ipm.2026.104878_b28","article-title":"Holistic evaluation of language models","author":"Liang","year":"2023","journal-title":"Transactions on Machine Learning Research"},{"issue":"11","key":"10.1016\/j.ipm.2026.104878_b29","doi-asserted-by":"crossref","DOI":"10.3390\/math13111707","article-title":"AI reasoning in deep learning era: From symbolic AI to neural\u2013symbolic AI","volume":"13","author":"Liang","year":"2025","journal-title":"Mathematics"},{"key":"10.1016\/j.ipm.2026.104878_b30","series-title":"The twelfth international conference on learning representations","article-title":"Let\u2019s verify step by step","author":"Lightman","year":"2024"},{"issue":"5","key":"10.1016\/j.ipm.2026.104878_b31","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2024.103809","article-title":"Are LLMs good at structured outputs? A benchmark for evaluating structured output capabilities in LLMs","volume":"61","author":"Liu","year":"2024","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.104878_b32","doi-asserted-by":"crossref","first-page":"2507","DOI":"10.52202\/068431-0182","article-title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","volume":"35","author":"Lu","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.ipm.2026.104878_b33","series-title":"Mixtral of experts","author":"Mistral AI","year":"2023"},{"key":"10.1016\/j.ipm.2026.104878_b34","series-title":"Proceedings of the 17th conference of the European chapter of the association for computational linguistics","first-page":"2014","article-title":"MTEB: Massive text embedding benchmark","author":"Muennighoff","year":"2023"},{"key":"10.1016\/j.ipm.2026.104878_b35","series-title":"Proceedings of the 41st international conference on machine learning","article-title":"PIVOT: iterative visual prompting elicits actionable knowledge for VLMs","author":"Nasiriany","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b36","doi-asserted-by":"crossref","first-page":"27730","DOI":"10.52202\/068431-2011","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.ipm.2026.104878_b37","series-title":"Advances in neural information processing systems","first-page":"70926","article-title":"Why think step by step? Reasoning emerges from the locality of experience","volume":"vol. 36","author":"Prystawski","year":"2023"},{"issue":"6","key":"10.1016\/j.ipm.2026.104878_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2023.103510","article-title":"What is the limitation of multimodal LLMs? A deeper look into multimodal LLMs through prompt probing","volume":"60","author":"Qi","year":"2023","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.ipm.2026.104878_b39","doi-asserted-by":"crossref","unstructured":"Reimers, N. Gurevych, I. (2019). Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. In Proceedings of the 2019 conference on empirical methods in natural language processing (pp. 3982\u20133992).","DOI":"10.18653\/v1\/D19-1410"},{"key":"10.1016\/j.ipm.2026.104878_b40","article-title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models","author":"Srivastava","year":"2023","journal-title":"Transactions on Machine Learning Research"},{"key":"10.1016\/j.ipm.2026.104878_b41","series-title":"Gemma 2: Improving open language models at a practical size","author":"Team","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b42","series-title":"Advances in neural information processing systems","first-page":"5998","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.ipm.2026.104878_b43","series-title":"Proceedings of the 41st international conference on machine learning","first-page":"50622","article-title":"SciBench: Evaluating college-level scientific problem-solving abilities of large language models","volume":"vol. 235","author":"Wang","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b44","series-title":"Advances in neural information processing systems","article-title":"SuperGLUE: A stickier benchmark for general-purpose language understanding systems","volume":"vol. 32","author":"Wang","year":"2019"},{"key":"10.1016\/j.ipm.2026.104878_b45","doi-asserted-by":"crossref","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O. Bowman, S. R. (2018). GLUE: A multi-task benchmark and analysis platform for natural language understanding. In Proceedings of the 2018 EMNLP workshop blackboxNLP (pp. 353\u2013355).","DOI":"10.18653\/v1\/W18-5446"},{"key":"10.1016\/j.ipm.2026.104878_b46","series-title":"The eleventh international conference on learning representations","article-title":"Self-consistency improves chain of thought reasoning in language models","author":"Wang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104878_b47","series-title":"Advances in neural information processing systems","first-page":"107403","article-title":"GFT: Graph foundation model with transferable tree vocabulary","volume":"vol. 37","author":"Wang","year":"2024"},{"key":"10.1016\/j.ipm.2026.104878_b48","unstructured":"Wang, Z., Zhang, Z., Ma, T., Chawla, N. V., Zhang, C., & Ye, Y. (2025). Beyond Message Passing: Neural Graph Pattern Machine. In Forty-second international conference on machine learning."},{"key":"10.1016\/j.ipm.2026.104878_b49","unstructured":"Wang, Z., Zhang, Z., Ma, T., Zhang, C., & Ye, Y. (2025). Generative Graph Pattern Machine. In The thirty-ninth annual conference on neural information processing systems."},{"key":"10.1016\/j.ipm.2026.104878_b50","article-title":"Emergent abilities of large language models","author":"Wei","year":"2022","journal-title":"Transactions on Machine Learning Research"},{"key":"10.1016\/j.ipm.2026.104878_b51","series-title":"Proceedings of the 36th international conference on neural information processing systems","article-title":"Chain-of-thought prompting elicits reasoning in large language models","author":"Wei","year":"2022"},{"issue":"12","key":"10.1016\/j.ipm.2026.104878_b52","doi-asserted-by":"crossref","first-page":"10272","DOI":"10.1109\/TPAMI.2024.3435448","article-title":"A diffusion model translator for efficient image-to-image translation","volume":"46","author":"Xia","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"6","key":"10.1016\/j.ipm.2026.104878_b53","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2025.104241","article-title":"DelphiAgent: A trustworthy multi-agent verification framework for automated fact verification","volume":"62","author":"Xiong","year":"2025","journal-title":"Information Processing & Management"},{"issue":"4","key":"10.1016\/j.ipm.2026.104878_b54","doi-asserted-by":"crossref","DOI":"10.1145\/3626235","article-title":"Diffusion models: A comprehensive survey of methods and applications","volume":"56","author":"Yang","year":"2023","journal-title":"ACM Computing Surveys"},{"issue":"3","key":"10.1016\/j.ipm.2026.104878_b55","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2024.104052","article-title":"Adaptive-solver framework for dynamic strategy selection in large language model reasoning","volume":"62","author":"Zhou","year":"2025","journal-title":"Information Processing & Management"}],"container-title":["Information Processing &amp; Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0306457326002694?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0306457326002694?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:39:16Z","timestamp":1783694356000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0306457326002694"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":55,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,11]]}},"alternative-id":["S0306457326002694"],"URL":"https:\/\/doi.org\/10.1016\/j.ipm.2026.104878","relation":{},"ISSN":["0306-4573"],"issn-type":[{"value":"0306-4573","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Cross-platform evaluation of reasoning capabilities in foundation models","name":"articletitle","label":"Article Title"},{"value":"Information Processing & Management","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.ipm.2026.104878","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104878"}}