{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T06:59:44Z","timestamp":1768978784386,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681294","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"6929-6938","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":15,"title":["mPLUG-PaperOwl: Scientific Diagram Analysis with the Multimodal Large Language Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8839-4996","authenticated-orcid":false,"given":"Anwen","family":"Hu","sequence":"first","affiliation":[{"name":"Alibaba Group, Chaoyang Qu, Beijing Shi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0465-6712","authenticated-orcid":false,"given":"Yaya","family":"Shi","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, Anhui, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9442-5912","authenticated-orcid":false,"given":"Haiyang","family":"Xu","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5451-8984","authenticated-orcid":false,"given":"Jiabo","family":"Ye","sequence":"additional","affiliation":[{"name":"East China Normal University, Putuo Qu, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7977-5540","authenticated-orcid":false,"given":"Qinghao","family":"Ye","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4959-8878","authenticated-orcid":false,"given":"Ming","family":"Yan","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9077-3928","authenticated-orcid":false,"given":"Chenliang","family":"Li","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8661-1169","authenticated-orcid":false,"given":"Qi","family":"Qian","sequence":"additional","affiliation":[{"name":"Alibaba Group, Bellevue, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3835-7975","authenticated-orcid":false,"given":"Ji","family":"Zhang","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3709-5053","authenticated-orcid":false,"given":"Fei","family":"Huang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Bellevue, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Radev","author":"Abu-Jbara Amjad","year":"2011","unstructured":"Amjad Abu-Jbara and Dragomir R. Radev. 2011. Coherent Citation-Based Summarization of Scientific Papers. In ACL. The Association for Computer Linguistics, 500--509."},{"key":"e_1_3_2_1_2_1","volume-title":"Flamingo: a Visual Language Model for Few-Shot Learning. ArXiv","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katie Millican, Malcolm Reynolds, Roman Ring, Eliza Rutherford, Serkan Cabi, Tengda Han, Zhitao Gong, Sina Samangooei, Marianne Monteiro, Jacob Menick, Sebastian Borgeaud, Andy Brock, Aida Nematzadeh, Sahand Sharifzadeh, Mikolaj Binkowski, Ricardo Barreira, Oriol Vinyals, Andrew Zisserman, and Karen Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. ArXiv, Vol. abs\/2204.14198 (2022)."},{"key":"e_1_3_2_1_3_1","volume-title":"Sam Skjonsberg, Lucy Lu Wang, Chris Wilhelm, Zheng Yuan, Madeleine van Zuylen, and Oren Etzioni.","author":"Ammar Waleed","year":"2018","unstructured":"Waleed Ammar, Dirk Groeneveld, Chandra Bhagavatula, Iz Beltagy, Miles Crawford, Doug Downey, Jason Dunkelberger, Ahmed Elgohary, Sergey Feldman, Vu Ha, Rodney Kinney, Sebastian Kohlmeier, Kyle Lo, Tyler Murray, Hsu-Han Ooi, Matthew E. Peters, Joanna Power, Sam Skjonsberg, Lucy Lu Wang, Chris Wilhelm, Zheng Yuan, Madeleine van Zuylen, and Oren Etzioni. 2018. Construction of the Literature Graph in Semantic Scholar. In NAACL-HLT (3). Association for Computational Linguistics, 84--91."},{"key":"e_1_3_2_1_4_1","volume-title":"Enhancing Scientific Papers Summarization with Citation Graph","author":"An Chenxin","unstructured":"Chenxin An, Ming Zhong, Yiran Chen, Danqing Wang, Xipeng Qiu, and Xuanjing Huang. 2021. Enhancing Scientific Papers Summarization with Citation Graph. In AAAI. AAAI Press, 12498--12506."},{"key":"e_1_3_2_1_5_1","volume-title":"Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. CoRR","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. CoRR, Vol. abs\/2308.12966 (2023)."},{"key":"e_1_3_2_1_6_1","volume-title":"METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In IEEvaluation@ACL","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In IEEvaluation@ACL. Association for Computational Linguistics, 65--72."},{"key":"e_1_3_2_1_7_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.1986.4767851"},{"key":"e_1_3_2_1_9_1","volume-title":"TabFact: A Large-scale Dataset for Table-based Fact Verification. In International Conference on Learning Representations (ICLR). Addis Ababa, Ethiopia.","author":"Chen Wenhu","year":"2020","unstructured":"Wenhu Chen, Hongmin Wang, Jianshu Chen, Yunkai Zhang, Hong Wang, Shiyang Li, Xiyou Zhou, and William Yang Wang. 2020. TabFact: A Large-scale Dataset for Table-based Fact Verification. In International Conference on Learning Representations (ICLR). Addis Ababa, Ethiopia."},{"key":"e_1_3_2_1_10_1","volume-title":"Junqi Zhao, Weisheng Wang, Boyang Li, Pascale Fung, and Steven C. H. Hoi.","author":"Dai Wenliang","year":"2023","unstructured":"Wenliang Dai, Junnan Li, Dongxu Li, Anthony Meng Huat Tiong, Junqi Zhao, Weisheng Wang, Boyang Li, Pascale Fung, and Steven C. H. Hoi. 2023. InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning. CoRR, Vol. abs\/2305.06500 (2023)."},{"key":"e_1_3_2_1_11_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly Jakob Uszkoreit and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In ICLR. OpenReview.net."},{"key":"e_1_3_2_1_12_1","volume-title":"Unidoc: A universal large multimodal model for simultaneous text detection, recognition, spotting and understanding. arXiv preprint arXiv:2308.11592","author":"Feng Hao","year":"2023","unstructured":"Hao Feng, Zijian Wang, Jingqun Tang, Jinghui Lu, Wengang Zhou, Houqiang Li, and Can Huang. 2023. Unidoc: A universal large multimodal model for simultaneous text detection, recognition, spotting and understanding. arXiv preprint arXiv:2308.11592 (2023)."},{"key":"e_1_3_2_1_13_1","volume-title":"ChartLlama: A Multimodal LLM for Chart Understanding and Generation. CoRR","author":"Han Yucheng","year":"2023","unstructured":"Yucheng Han, Chi Zhang, Xin Chen, Xu Yang, Zhibin Wang, Gang Yu, Bin Fu, and Hanwang Zhang. 2023. ChartLlama: A Multimodal LLM for Chart Understanding and Generation. CoRR, Vol. abs\/2311.16483 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"CogAgent: A Visual Language Model for GUI Agents. CoRR","author":"Hong Wenyi","year":"2023","unstructured":"Wenyi Hong, Weihan Wang, Qingsong Lv, Jiazheng Xu, Wenmeng Yu, Junhui Ji, Yan Wang, Zihan Wang, Yuxuan Zhang, Juanzi Li, Bin Xu, Yuxiao Dong, Ming Ding, and Jie Tang. 2023. CogAgent: A Visual Language Model for GUI Agents. CoRR, Vol. abs\/2312.08914 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"EMNLP (Findings)","author":"Hsu Ting-Yao","unstructured":"Ting-Yao Hsu, C. Lee Giles, and Ting-Hao Kenneth Huang. 2021. SciCap: Generating Captions for Scientific Figures. In EMNLP (Findings). Association for Computational Linguistics, 3258--3264."},{"key":"e_1_3_2_1_16_1","volume-title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding. CoRR","author":"Hu Anwen","year":"2024","unstructured":"Anwen Hu, Haiyang Xu, Jiabo Ye, Ming Yan, Liang Zhang, Bo Zhang, Chen Li, Ji Zhang, Qin Jin, Fei Huang, and Jingren Zhou. 2024. mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding. CoRR, Vol. abs\/2403.12895 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_2_1_18_1","volume-title":"DVQA: Understanding Data Visualizations via Question Answering","author":"Kafle Kushal","year":"2018","unstructured":"Kushal Kafle, Brian L. Price, Scott Cohen, and Christopher Kanan. 2018. DVQA: Understanding Data Visualizations via Question Answering. In CVPR. Computer Vision Foundation \/ IEEE Computer Society, 5648--5656."},{"key":"e_1_3_2_1_19_1","unstructured":"Samira Ebrahimi Kahou Vincent Michalski Adam Atkinson \u00c1kos K\u00e1d\u00e1r Adam Trischler and Yoshua Bengio. 2018. FigureQA: An Annotated Figure Dataset for Visual Reasoning. In ICLR (Workshop). OpenReview.net."},{"key":"e_1_3_2_1_20_1","volume-title":"Xiang Lin, Ahmed Masry, Megh Thakkar, Enamul Hoque, and Shafiq R. Joty.","author":"Kantharaj Shankar","year":"2022","unstructured":"Shankar Kantharaj, Rixie Tiffany Ko Leong, Xiang Lin, Ahmed Masry, Megh Thakkar, Enamul Hoque, and Shafiq R. Joty. 2022. Chart-to-Text: A Large-Scale Benchmark for Chart Summarization. In ACL (1). Association for Computational Linguistics, 4005--4023."},{"key":"e_1_3_2_1_21_1","volume-title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs. CoRR","author":"Li Shengzhi","year":"2023","unstructured":"Shengzhi Li and Nima Tajbakhsh. 2023. SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs. CoRR, Vol. abs\/2308.03349 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74--81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74--81."},{"key":"e_1_3_2_1_23_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li and Yong Jae Lee. 2023. Improved Baselines with Visual Instruction Tuning."},{"key":"e_1_3_2_1_24_1","volume-title":"Visual Instruction Tuning. CoRR","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual Instruction Tuning. CoRR, Vol. abs\/2304.08485 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document. CoRR","author":"Liu Yuliang","year":"2024","unstructured":"Yuliang Liu, Biao Yang, Qiang Liu, Zhang Li, Zhiyin Ma, Shuo Zhang, and Xiang Bai. 2024. TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document. CoRR, Vol. abs\/2403.04473 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"Mark Neumann, Rodney Kinney, and Daniel S. Weld.","author":"Lo Kyle","year":"2020","unstructured":"Kyle Lo, Lucy Lu Wang, Mark Neumann, Rodney Kinney, and Daniel S. Weld. 2020. S2ORC: The Semantic Scholar Open Research Corpus. In ACL. Association for Computational Linguistics, 4969--4983."},{"key":"e_1_3_2_1_27_1","volume-title":"Jia Qing Tan, Shafiq R. Joty, and Enamul Hoque.","author":"Masry Ahmed","year":"2022","unstructured":"Ahmed Masry, Do Xuan Long, Jia Qing Tan, Shafiq R. Joty, and Enamul Hoque. 2022. ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning. In ACL (Findings). Association for Computational Linguistics, 2263--2279."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Minesh Mathew Viraj Bagal Rub\u00e8n Tito Dimosthenis Karatzas Ernest Valveny and C. V. Jawahar. 2022. InfographicVQA. In WACV. IEEE 2582--2591.","DOI":"10.1109\/WACV51458.2022.00264"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Minesh Mathew Dimosthenis Karatzas and C. V. Jawahar. 2021. DocVQA: A Dataset for VQA on Document Images. In WACV. IEEE 2199--2208.","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"e_1_3_2_1_30_1","volume-title":"ChartAssisstant: A Universal Chart Multimodal Language Model via Chart-to-Table Pre-training and Multitask Instruction Tuning. CoRR","author":"Meng Fanqing","year":"2024","unstructured":"Fanqing Meng, Wenqi Shao, Quanfeng Lu, Peng Gao, Kaipeng Zhang, Yu Qiao, and Ping Luo. 2024. ChartAssisstant: A Universal Chart Multimodal Language Model via Chart-to-Table Pre-training and Multitask Instruction Tuning. CoRR, Vol. abs\/2401.02384 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"PlotQA: Reasoning over Scientific Plots","author":"Methani Nitesh","unstructured":"Nitesh Methani, Pritha Ganguly, Mitesh M. Khapra, and Pratyush Kumar. 2020. PlotQA: Reasoning over Scientific Plots. In WACV. IEEE, 1516--1525."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318."},{"key":"e_1_3_2_1_33_1","volume-title":"ACL (1)","author":"Pasupat Panupong","unstructured":"Panupong Pasupat and Percy Liang. 2015. Compositional Semantic Parsing on Semi-Structured Tables. In ACL (1). The Association for Computer Linguistics, 1470--1480."},{"key":"e_1_3_2_1_34_1","volume-title":"BIR@ECIR (CEUR Workshop Proceedings","author":"Saier Tarek","unstructured":"Tarek Saier and Michael F\u00e4rber. 2019. Bibliometric-Enhanced arXiv: A Data Set for Paper-Based and Citation-Based Tasks. In BIR@ECIR (CEUR Workshop Proceedings, Vol. 2345). CEUR-WS.org, 14--26."},{"key":"e_1_3_2_1_35_1","volume-title":"ACL (4)","author":"Shen Zhihong","unstructured":"Zhihong Shen, Hao Ma, and Kuansan Wang. 2018. A Web-scale system for scientific knowledge exploration. In ACL (4). Association for Computational Linguistics, 87--92."},{"key":"e_1_3_2_1_36_1","volume-title":"Kleister: Key Information Extraction Datasets Involving Long Documents with Complex Layouts. In ICDAR (1) (Lecture Notes in Computer Science","author":"Stanislawek Tomasz","year":"2021","unstructured":"Tomasz Stanislawek, Filip Gralinski, Anna Wr\u00f3blewska, Dawid Lipinski, Agnieszka Kaliska, Paulina Rosalska, Bartosz Topolski, and Przemyslaw Biecek. 2021. Kleister: Key Information Extraction Datasets Involving Long Documents with Complex Layouts. In ICDAR (1) (Lecture Notes in Computer Science, Vol. 12821). Springer, 564--579."},{"key":"e_1_3_2_1_37_1","unstructured":"S Svetlichnaya. 2020. DeepForm: Understand structured documents at scale."},{"key":"e_1_3_2_1_38_1","volume-title":"ACL (1)","author":"Tang Benny J.","unstructured":"Benny J. Tang, Angie Boggust, and Arvind Satyanarayan. 2023. VisText: A Benchmark for Semantically Rich Chart Captioning. In ACL (1). Association for Computational Linguistics, 7268--7298."},{"key":"e_1_3_2_1_39_1","volume-title":"Hashimoto","author":"Taori Rohan","year":"2023","unstructured":"Rohan Taori, Ishaan Gulrajani, Tianyi Zhang, Yann Dubois, Xuechen Li, Carlos Guestrin, Percy Liang, and Tatsunori B. Hashimoto. 2023. Stanford Alpaca: An Instruction-following LLaMA model. https:\/\/github.com\/tatsu-lab\/stanford_alpaca."},{"key":"e_1_3_2_1_40_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_42_1","volume-title":"CIDEr: Consensus-based image description evaluation","author":"Vedantam Ramakrishna","unstructured":"Ramakrishna Vedantam, C. Lawrence Zitnick, and Devi Parikh. 2015. CIDEr: Consensus-based image description evaluation. In CVPR. IEEE Computer Society, 4566--4575."},{"key":"e_1_3_2_1_43_1","volume-title":"Vicuna: An Open Chatbot Impressing GPT-4. https:\/\/github.com\/lm-sys\/FastChat.","year":"2023","unstructured":"Vicuna. 2023. Vicuna: An Open Chatbot Impressing GPT-4. https:\/\/github.com\/lm-sys\/FastChat."},{"key":"e_1_3_2_1_44_1","volume-title":"CogVLM: Visual Expert for Pretrained Language Models. CoRR","author":"Wang Weihan","year":"2023","unstructured":"Weihan Wang, Qingsong Lv, Wenmeng Yu, Wenyi Hong, Ji Qi, Yan Wang, Junhui Ji, Zhuoyi Yang, Lei Zhao, Xixuan Song, Jiazheng Xu, Bin Xu, Juanzi Li, Yuxiao Dong, Ming Ding, and Jie Tang. 2023. CogVLM: Visual Expert for Pretrained Language Models. CoRR, Vol. abs\/2311.03079 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Towards Improving Document Understanding: An Exploration on Text-Grounding via MLLMs. arXiv preprint arXiv:2311.13194","author":"Wang Yonghui","year":"2023","unstructured":"Yonghui Wang, Wengang Zhou, Hao Feng, Keyi Zhou, and Houqiang Li. 2023. Towards Improving Document Understanding: An Exploration on Text-Grounding via MLLMs. arXiv preprint arXiv:2311.13194 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"A Versatile Benchmark and Foundation Model for Complicated Chart Reasoning. CoRR","author":"Xia Renqiu","year":"2024","unstructured":"Renqiu Xia, Bo Zhang, Hancheng Ye, Xiangchao Yan, Qi Liu, Hongbin Zhou, Zijun Chen, Min Dou, Botian Shi, Junchi Yan, and Yu Qiao. 2024. ChartX & ChartVLM: A Versatile Benchmark and Foundation Model for Complicated Chart Reasoning. CoRR, Vol. abs\/2402.12185 (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"SciCap: A Knowledge Augmented Dataset to Study the Challenges of Scientific Figure Captioning. CoRR","author":"Yang Zhishen","year":"2023","unstructured":"Zhishen Yang, Raj Dabre, Hideki Tanaka, and Naoaki Okazaki. 2023. SciCap: A Knowledge Augmented Dataset to Study the Challenges of Scientific Figure Captioning. CoRR, Vol. abs\/2306.03491 (2023)."},{"key":"e_1_3_2_1_48_1","volume-title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding. CoRR","author":"Ye Jiabo","year":"2023","unstructured":"Jiabo Ye, Anwen Hu, Haiyang Xu, Qinghao Ye, Ming Yan, Yuhao Dan, Chenlin Zhao, Guohai Xu, Chenliang Li, Junfeng Tian, Qian Qi, Ji Zhang, and Fei Huang. 2023. mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding. CoRR, Vol. abs\/2307.02499 (2023)."},{"key":"e_1_3_2_1_49_1","volume-title":"Xin Alex Lin, and Fei Huang","author":"Ye Jiabo","year":"2023","unstructured":"Jiabo Ye, Anwen Hu, Haiyang Xu, Qinghao Ye, Ming Yan, Guohai Xu, Chenliang Li, Junfeng Tian, Qi Qian, Ji Zhang, Qin Jin, Liang He, Xin Alex Lin, and Fei Huang. 2023. UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model. CoRR, Vol. abs\/2310.05126 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality. CoRR","author":"Ye Qinghao","year":"2023","unstructured":"Qinghao Ye, Haiyang Xu, Guohai Xu, Jiabo Ye, Ming Yan, Yiyang Zhou, Junyang Wang, Anwen Hu, Pengcheng Shi, Yaya Shi, Chenliang Li, Yuanhong Xu, Hehong Chen, Junfeng Tian, Qian Qi, Ji Zhang, and Fei Huang. 2023. mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality. CoRR, Vol. abs\/2304.14178 (2023)."},{"key":"e_1_3_2_1_51_1","unstructured":"Qinghao Ye Haiyang Xu Jiabo Ye Ming Yan Anwen Hu Haowei Liu Qi Qian Ji Zhang Fei Huang and Jingren Zhou. 2023 d. mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration. arxiv: 2311.04257 [cs.CL]"},{"key":"e_1_3_2_1_52_1","volume-title":"MPMQA: Multimodal Question Answering on Product Manuals","author":"Zhang Liang","year":"2023","unstructured":"Liang Zhang, Anwen Hu, Jing Zhang, Shuo Hu, and Qin Jin. 2023. MPMQA: Multimodal Question Answering on Product Manuals. In AAAI. AAAI Press, 13958--13966."},{"key":"e_1_3_2_1_53_1","volume-title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding. CoRR","author":"Zhang Yanzhe","year":"2023","unstructured":"Yanzhe Zhang, Ruiyi Zhang, Jiuxiang Gu, Yufan Zhou, Nedim Lipka, Diyi Yang, and Tong Sun. 2023. LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding. CoRR, Vol. abs\/2306.17107 (2023)."},{"key":"e_1_3_2_1_54_1","unstructured":"Deyao Zhu Jun Chen Xiaoqian Shen Xiang Li and Mohamed Elhoseiny. 2023. MiniGPT-4: Enhancing Vision-language Understanding with Advanced Large Language Models."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681294","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681294","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:43Z","timestamp":1750295863000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681294"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":54,"alternative-id":["10.1145\/3664647.3681294","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681294","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}