{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,2]],"date-time":"2026-03-02T22:11:57Z","timestamp":1772489517225,"version":"3.50.1"},"reference-count":137,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1109\/tai.2025.3606452","type":"journal-article","created":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T17:50:37Z","timestamp":1757353837000},"page":"1715-1729","source":"Crossref","is-referenced-by-count":0,"title":["Dual Thinking and Logical Processing in Human Vision and Multimodal Large Language Models"],"prefix":"10.1109","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6472-0662","authenticated-orcid":false,"given":"Kailas","family":"Dayanandan","sequence":"first","affiliation":[{"name":"Indian Institute of Technology, New Delhi, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5643-7180","authenticated-orcid":false,"given":"Nikhil","family":"Kumar","sequence":"additional","affiliation":[{"name":"Indraprastha Institute of Information Technology Delhi, New Delhi, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4025-359X","authenticated-orcid":false,"given":"Anand","family":"Sinha","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology, New Delhi, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2677-3071","authenticated-orcid":false,"given":"Brejesh","family":"Lall","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology, New Delhi, India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.2478\/v10053-008-0022-3"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.conb.2020.11.009"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.7554\/eLife.36329"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/3152042.3152052"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87199-4_14"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuroimage.2018.12.046"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1111\/nyas.14320"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/381520a0"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1719397115"},{"key":"ref10","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1037\/a0029333"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.3758\/s13423-023-02344-9"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1037\/a0024330"},{"key":"ref14","article-title":"Experimentelle studien uber das sehen von bewegung","volume":"61","author":"Wertheimer","year":"1912","journal-title":"Zeitschrift Psychologie"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1037\/0096-1523.3.3.422"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1037\/a0029334"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s42113-023-00169-2"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/s42113-021-00100-7"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1111\/tops.12136"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1111\/tops.12131"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1111\/tops.12137"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/T-C.1971.223083"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2008.2002306"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2405350"},{"key":"ref25","article-title":"Improving neural network representations using human similarity judgments","volume-title":"Proc. 37th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Muttenthaler","year":"2024"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-020-00257-z"},{"key":"ref27","first-page":"13073","article-title":"Simulating a primary visual cortex at the front of CNNs improves robustness to image perturbations","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Dapello","year":"2020"},{"key":"ref28","article-title":"Learning what and where to attend","volume-title":"Int. Conf. Learn. Representations","author":"Linsley","year":"2019"},{"key":"ref29","article-title":"Stable and expressive recurrent vision models","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Linsley","year":"2020"},{"key":"ref30","first-page":"9432","article-title":"Harmonizing the object recognition strategies of deep neural networks with humans","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Fel","year":"2022"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1800901115"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1038\/s41583-020-00395-8"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.32470\/CCN.2019.1130-0"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1017\/S0140525X16001837"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1017\/s0140525x22002813"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-vision-091718-014951"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pcbi.1006613"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1811.12231"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"ref40","first-page":"88527","article-title":"Hidden in plain sight: Evaluating abstract shape recognition in vision-language models","volume-title":"Proc. 38th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Hemmat","year":"2025"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00247"},{"key":"ref43","article-title":"The origins and prevalence of texture bias in convolutional neural networks","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Hermann","year":"2020"},{"key":"ref44","first-page":"13890","article-title":"Beyond accuracy: Quantifying trial-by-trial behaviour of CNNs and humans by measuring error consistency","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Geirhos","year":"2020"},{"key":"ref45","first-page":"23885","article-title":"Partial success in closing the gap between human and machine vision","volume-title":"Proc. 35th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Geirhos","year":"2021"},{"key":"ref46","first-page":"7549","article-title":"Generalisation in humans and deep neural networks","volume-title":"Proc. 32nd Annu. Conf. Neural Inf. Process. Syst. (NeurIPS)","author":"Geirhos","year":"2019"},{"issue":"43","key":"ref47","article-title":"Are convolutional neural networks or transformers more like human vision?","volume-title":"Proc. Annu. Meeting Cogn. Sci. Soc.","volume":"43","author":"Tuli","year":"2021"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pcbi.1003963"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-021-22078-3"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-020-19632-w"},{"key":"ref51","article-title":"Do vision transformers see like convolutional neural networks?","volume-title":"Proc. 35th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Raghu","year":"2021"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pcbi.1011280"},{"key":"ref53","article-title":"Intriguing properties of generative classifiers","volume-title":"12th Int. Conf. Learn. Representations","author":"Jaini","year":"2023"},{"key":"ref54","article-title":"Scaling LLM test-time compute optimally can be more effective than scaling model parameters","author":"Snell","year":"2024"},{"key":"ref55","article-title":"Robust and generalizable visual representation learning via random convolutions","volume-title":"Int. Conf. Learn. Representations","author":"Xu","year":"2021"},{"key":"ref56","article-title":"Shape-texture debiased neural network training","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li","year":"2021"},{"key":"ref57","article-title":"Adversarial examples are not bugs, they are features","volume-title":"Proc. 33rd Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Ilyas","year":"2019"},{"key":"ref58","article-title":"Spatial-frequency channels, shape bias, and adversarial robustness","volume-title":"Proc. 37th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Subramanian","year":"2024"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.00743"},{"key":"ref60","article-title":"Approximating CNNs with bag-of-local-features models works surprisingly well on ImageNet","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Brendel","year":"2018"},{"key":"ref61","article-title":"Does enhanced shape bias improve neural network robustness to common corruptions?","volume-title":"Int. Conf. Learn. Representations","author":"Mummadi","year":"2021"},{"key":"ref62","article-title":"Computational complexity of segmentation","volume-title":"Annu. Meeting Cogn. Sci. Soc.","author":"Adolfi","year":"2022"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1111\/tops.12629"},{"key":"ref64","doi-asserted-by":"crossref","DOI":"10.1016\/j.heares.2020.107998","article-title":"Active listening","volume":"399","author":"Friston","year":"2021","journal-title":"Hearing Res."},{"key":"ref65","article-title":"Diffusion models as artists: Are we closing the gap between humans and machines?","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Boutin","year":"2023"},{"key":"ref66","first-page":"20933","article-title":"Diversity vs. recognizability: Human-like generalization in one-shot generative models","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Boutin","year":"2022"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-vision-093019-111701"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3609224"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25087"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00498"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2018.00212"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00040"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00019"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1145\/3636551"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00285"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3130490"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3085766"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01142"},{"issue":"6","key":"ref81","first-page":"7","article-title":"Animal camouflage analysis: Chameleon database","volume":"2","author":"Skurowski","year":"2018"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2009.5459462"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1015666108"},{"key":"ref85","article-title":"Scaling MLPs: A tale of inductive bias","volume-title":"Proc. 37th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Bachmann","year":"2024"},{"key":"ref86","article-title":"ViTAE: Vision transformer advanced by exploring intrinsic inductive bias","volume-title":"Proc. 35th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Xu","year":"2021"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650805"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00686"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00331"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00439"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00145"},{"key":"ref93","article-title":"Measuring multimodal mathematical reasoning with math-vision dataset","author":"Wang","year":"2024"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.356"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1241"},{"key":"ref97","article-title":"Detect, describe, discriminate: Moving beyond VQA for MLLM evaluation","author":"Gaur","year":"2024"},{"key":"ref98","article-title":"KiVA: Kid-inspired visual analogies for testing large multimodal models","author":"Yiu","year":"2024"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3604"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1037\/rev0000177"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1016\/S1364-6613(99)01350-9"},{"key":"ref102","article-title":"Spurious equilibrium in segmentation models and recurrent processing in human vision","volume-title":"Workshop on Spurious Correlation Shortcut Learn.: Found. Solutions","author":"Kailas","year":"2025"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00223"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3159394"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01508"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1145\/3449287"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1016\/j.chest.2022.10.039"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1177\/2041669518788887"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1523\/JNEUROSCI.1996-04.2004"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1126\/science.287.5461.2115a"},{"key":"ref111","first-page":"25278","article-title":"LAION-5B: An open large-scale dataset for training next generation image-text models","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Schuhmann","year":"2022"},{"key":"ref112","article-title":"Gpt-4o mini: Advancing cost-efficient intelligence","year":"2024"},{"key":"ref113","article-title":"GPT-4o system card","author":"Hurst","year":"2024"},{"key":"ref114","article-title":"Llama 3.2: Revolutionizing edge AI and vision with open, customizable models","year":"2024"},{"key":"ref115","article-title":"Qwen2.5-VL technical report","author":"Bai","year":"2025"},{"key":"ref116","article-title":"Introducing Gemini 2.0: Our new AI model for the agentic era","author":"DeepMind","year":"2024"},{"key":"ref117","article-title":"Introducing Claude 4","year":"2025"},{"key":"ref118","article-title":"Molmo and PixMo: Open weights and open data for state-of-the-art multimodal models","author":"Deitke","year":"2024"},{"key":"ref119","article-title":"Pixtral 12B","author":"Agrawal","year":"2024"},{"key":"ref120","article-title":"MMDetection: Open MMLab detection toolbox and benchmark","author":"Chen","year":"2019"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00077"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00925"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412258"},{"key":"ref124","first-page":"10285","article-title":"A theoretical analysis of the learning dynamics under class imbalance","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Francazi","year":"2023"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01436"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1109\/CONIT59222.2023.10205557"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1038\/s41591-018-0268-3"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1038\/s41746-019-0146-5"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-018-07229-3"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1117\/1.JMI.10.4.044503"},{"key":"ref131","article-title":"Self-consistency improves chain of thought reasoning in language models","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Wang","year":"2023"},{"key":"ref132","article-title":"Large language monkeys: Scaling inference compute with repeated sampling","author":"Brown","year":"2024"},{"key":"ref133","article-title":"Large language model guided tree-of-thought","author":"Long","year":"2023"},{"key":"ref134","article-title":"Benchmarking and improving generator-validator consistency of language models","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Li","year":"2024"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.70777\/si.v2i6.15919"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-024-49194-0"},{"key":"ref137","article-title":"Training compute-optimal large language models","author":"Hoffmann","year":"2022"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/11417361\/11153039.pdf?arnumber=11153039","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,2]],"date-time":"2026-03-02T20:58:55Z","timestamp":1772485135000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11153039\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":137,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tai.2025.3606452","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]}}}