{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T14:02:29Z","timestamp":1785074549445,"version":"3.55.0"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100005047","name":"Liaoning Provincial Natural Science Foundation","doi-asserted-by":"publisher","award":["2024-MSBA-49"],"award-info":[{"award-number":["2024-MSBA-49"]}],"id":[{"id":"10.13039\/501100005047","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576083"],"award-info":[{"award-number":["62576083"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62432003"],"award-info":[{"award-number":["62432003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U25A20431"],"award-info":[{"award-number":["U25A20431"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116518","type":"journal-article","created":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:25:43Z","timestamp":1782318343000},"page":"116518","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Less is more: Mitigating attention bias in multimodal table understanding through table content erasure"],"prefix":"10.1016","volume":"349","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2946-7910","authenticated-orcid":false,"given":"Yuliang","family":"Liang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1709-5056","authenticated-orcid":false,"given":"Guibing","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0704-6621","authenticated-orcid":false,"given":"Wei","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianzhe","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xingwei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116518_b1","series-title":"NAACL","article-title":"Tablellama: Towards open large generalist models for tables","author":"Zhang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116518_b2","series-title":"ACL","article-title":"Multimodal table understanding","author":"Zheng","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b3","series-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b4","series-title":"CVPR","first-page":"24796","article-title":"SynTab-llava: Enhancing multimodal table understanding with decoupled synthesis","author":"Zhou","year":"2025"},{"key":"10.1016\/j.knosys.2026.116518_b5","article-title":"Chain-of-table: Evolving tables in the reasoning chain for table understanding","author":"Wang","year":"2024","journal-title":"ICLR"},{"key":"10.1016\/j.knosys.2026.116518_b6","series-title":"Tablegpt2: A large multimodal model with tabular data integration","author":"Su","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b7","series-title":"GPT-4 technical report","author":"OpenA.I. and many contributors","year":"2023"},{"key":"10.1016\/j.knosys.2026.116518_b8","series-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b9","series-title":"WSDM","article-title":"Table meets llm: Can large language models understand structured table data? a benchmark and empirical study","author":"Sui","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b10","series-title":"ACL","article-title":"PixT3: Pixel-based table to text generation","author":"Alonso","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b11","series-title":"NeurIPS","article-title":"Tabpedia: Towards comprehensive visual table understanding with concept synergy","author":"Zhao","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b12","series-title":"ACL","first-page":"407","article-title":"Tables as texts or images: Evaluating the table reasoning ability of llms and mllms","author":"Deng","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b13","series-title":"ACL","article-title":"Hitab: A hierarchical table dataset for question answering and natural language generation","author":"Cheng","year":"2022"},{"key":"10.1016\/j.knosys.2026.116518_b14","series-title":"CVPR","first-page":"13872","article-title":"Mitigating object hallucinations in large vision-language models through visual contrastive decoding","author":"Leng","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b15","series-title":"Mitigating modality prior-induced hallucinations in multimodal large language models via deciphering attention causality","author":"Zhou","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b16","series-title":"European Conference on Computer Vision","first-page":"198","article-title":"Contrastive region guidance: Improving grounding in vision-language models without training","author":"Wan","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b17","doi-asserted-by":"crossref","unstructured":"D. Xue, S. Qian, C. Xu, Variational causal inference network for explanatory visual question answering, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 2515\u20132525.","DOI":"10.1109\/ICCV51070.2023.00238"},{"issue":"12","key":"10.1016\/j.knosys.2026.116518_b18","doi-asserted-by":"crossref","first-page":"7893","DOI":"10.1109\/TPAMI.2024.3398012","article-title":"Integrating neural-symbolic reasoning with variational causal inference network for explanatory visual question answering","volume":"46","author":"Xue","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.knosys.2026.116518_b19","doi-asserted-by":"crossref","unstructured":"D. Xue, S. Qian, C. Xu, Few-shot multimodal explanation for visual question answering, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 1875\u20131884.","DOI":"10.1145\/3664647.3681597"},{"key":"10.1016\/j.knosys.2026.116518_b20","doi-asserted-by":"crossref","first-page":"16","DOI":"10.1109\/TMM.2024.3521709","article-title":"Linin: Logic integrated neural inference network for explanatory visual question answering","volume":"27","author":"Xue","year":"2024","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.knosys.2026.116518_b21","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"NeurIPS"},{"key":"10.1016\/j.knosys.2026.116518_b22","series-title":"ICLR","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"10.1016\/j.knosys.2026.116518_b23","series-title":"Qwen2.5 technical report","author":"Yang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b24","first-page":"2024","article-title":"Llama 3.2: Revolutionizing edge AI and vision with open, customizable models","volume":"20","author":"Meta","year":"2024","journal-title":"Meta AI Blog. Retrieved Dec."},{"key":"10.1016\/j.knosys.2026.116518_b25","series-title":"CVPR","first-page":"16000","article-title":"Masked autoencoders are scalable vision learners","author":"He","year":"2022"},{"key":"10.1016\/j.knosys.2026.116518_b26","series-title":"ACL","article-title":"Compositional semantic parsing on semi-structured tables","author":"Pasupat","year":"2015"},{"key":"10.1016\/j.knosys.2026.116518_b27","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1214\/09-SS057","article-title":"Causal inference in statistics: An overview","volume":"3","author":"Pearl","year":"2009","journal-title":"Stat. Surv."},{"key":"10.1016\/j.knosys.2026.116518_b28","series-title":"CVPR","first-page":"12700","article-title":"Counterfactual vqa: A cause-effect look at language bias","author":"Niu","year":"2021"},{"key":"10.1016\/j.knosys.2026.116518_b29","series-title":"NeurIPS","article-title":"Chain-of-thought prompting elicits reasoning in large language models","author":"Wei","year":"2022"},{"key":"10.1016\/j.knosys.2026.116518_b30","series-title":"ICLR 2023","article-title":"Self-consistency improves chain of thought reasoning in language models","author":"Wang","year":"2022"},{"key":"10.1016\/j.knosys.2026.116518_b31","series-title":"European Conference on Computer Vision","first-page":"125","article-title":"Paying more attention to image: A training-free method for alleviating hallucination in lvlms","author":"Liu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b32","series-title":"The hidden life of tokens: Reducing hallucination of large vision-language models via visual information steering","author":"Li","year":"2025"},{"key":"10.1016\/j.knosys.2026.116518_b33","series-title":"ICLR","article-title":"Dynamic prompt learning via policy gradient for semi-structured mathematical reasoning","author":"Lu","year":"2023"},{"key":"10.1016\/j.knosys.2026.116518_b34","series-title":"ACL","article-title":"TAT-QA: A question answering benchmark on a hybrid of tabular and textual content in finance","author":"Zhu","year":"2021"},{"key":"10.1016\/j.knosys.2026.116518_b35","series-title":"ECCV","first-page":"19","article-title":"An image is worth 1\/2 tokens after layer 2: Plug-and-play inference acceleration for large vision-language models","author":"Chen","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b36","series-title":"CVPR","first-page":"19792","article-title":"Visionzip: Longer is better but not necessary in vision language models","author":"Yang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116518_b37","series-title":"PruneVid: Visual token pruning for efficient video large language models","author":"Huang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b38","series-title":"ECCV","first-page":"214","article-title":"Ivtp: Instruction-guided visual token pruning for large vision-language models","author":"Huang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b39","series-title":"AAAI","first-page":"22128","article-title":"Fit and prune: Fast and training-free visual token pruning for multi-modal large language models","volume":"vol. 39","author":"Ye","year":"2025"},{"key":"10.1016\/j.knosys.2026.116518_b40","series-title":"ICLR","article-title":"Binding language models in symbolic languages","author":"Cheng","year":"2023"},{"key":"10.1016\/j.knosys.2026.116518_b41","series-title":"SIGIR","article-title":"Large language models are versatile decomposers: Decomposing evidence and questions for table-based reasoning","author":"Ye","year":"2023"},{"key":"10.1016\/j.knosys.2026.116518_b42","series-title":"Tree-of-table: Unleashing the power of LLMs for enhanced large-scale table understanding","author":"Ji","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b43","series-title":"Tablegpt: Towards unifying tables, nature language and commands into one gpt","author":"Zha","year":"2023"},{"key":"10.1016\/j.knosys.2026.116518_b44","series-title":"ICAIF","article-title":"Tat-llm: A specialized language model for discrete reasoning over tabular and textual data","author":"Zhu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b45","series-title":"TableLLM: Enabling tabular data manipulation by LLMs in real office usage scenarios","author":"Zhang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b46","article-title":"Table-GPT: Table fine-tuned GPT for diverse table tasks","author":"Li","year":"2024","journal-title":"PACMMOD"},{"key":"10.1016\/j.knosys.2026.116518_b47","series-title":"Tree-of-table: Unleashing the power of llms for enhanced large-scale table understanding","author":"Ji","year":"2024"},{"key":"10.1016\/j.knosys.2026.116518_b48","unstructured":"Z. Yang, Z. Du, M. Zhang, W. Du, J. Chen, Z. Duan, S. Zhao, Triples as the key: Structuring makes decomposition and verification easier in LLM-based TableQA, in: The Thirteenth International Conference on Learning Representations, 2025."}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S095070512601244X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S095070512601244X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T13:34:41Z","timestamp":1785072881000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S095070512601244X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":48,"alternative-id":["S095070512601244X"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116518","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Less is more: Mitigating attention bias in multimodal table understanding through table content erasure","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116518","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116518"}}