{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:04:12Z","timestamp":1765339452266,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206267"],"award-info":[{"award-number":["62206267"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"China Postdoctoral Science Foundation Funded Project","award":["2024M763867"],"award-info":[{"award-number":["2024M763867"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755397","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:54:15Z","timestamp":1761375255000},"page":"4349-4358","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["U-MERE: Unconstrained Multimodal Entity and Relation Extraction with Collaborative Modeling and Order-Sensitive Optimization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-5752-2132","authenticated-orcid":false,"given":"Wei","family":"Jia","sequence":"first","affiliation":[{"name":"Key Laboratory of Target Cognition and Application Technology, Aerospace Information Research Institute, Chinese Academy of Sciences, Beijing, China and School of Electronic, Electrical and Communication Engineering, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8833-4862","authenticated-orcid":false,"given":"Li","family":"Jin","sequence":"additional","affiliation":[{"name":"Key Laboratory of Target Cognition and Application Technology, Aerospace Information Research Institute, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5830-0802","authenticated-orcid":false,"given":"Kaiwen","family":"Wei","sequence":"additional","affiliation":[{"name":"College of Computer Science, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8052-2918","authenticated-orcid":false,"given":"Yuying","family":"Shang","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7664-9856","authenticated-orcid":false,"given":"Nayu","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Tiangong University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7520-7799","authenticated-orcid":false,"given":"Zhicong","family":"Lu","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0592-171X","authenticated-orcid":false,"given":"Qing","family":"Liu","sequence":"additional","affiliation":[{"name":"Aerospace Information Research Institute, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2631-0555","authenticated-orcid":false,"given":"Linhao","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5169-4634","authenticated-orcid":false,"given":"Jiang","family":"Zhong","sequence":"additional","affiliation":[{"name":"College of Computer Science, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0119-126X","authenticated-orcid":false,"given":"Yanfeng","family":"Hu","sequence":"additional","affiliation":[{"name":"Aerospace Information Research Institute, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. CoRR","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. CoRR, Vol. abs\/2308.12966 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"abs\/2502.13923","author":"Bai Shuai","year":"2025","unstructured":"Shuai Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Sibo Song, Kai Dang, Peng Wang, Shijie Wang, Jun Tang, Humen Zhong, Yuanzhi Zhu, Ming-Hsuan Yang, Zhaohai Li, Jianqiang Wan, Pengfei Wang, Wei Ding, Zheren Fu, Yiheng Xu, Jiabo Ye, Xi Zhang, Tianbao Xie, Zesen Cheng, Hang Zhang, Zhibo Yang, Haiyang Xu, and Junyang Lin. 2025. Qwen2.5-VL Technical Report. CoRR, Vol. abs\/2502.13923 (2025)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.4"},{"key":"e_1_3_2_1_4_1","first-page":"2969","volume-title":"In-context Learning for Few-shot Multimodal Named Entity Recognition. In Findings of the Association for Computational Linguistics: EMNLP","author":"Cai Chenran","year":"2023","unstructured":"Chenran Cai, Qianlong Wang, Bin Liang, Bing Qin, Min Yang, Kam-Fai Wong, and Ruifeng Xu. 2023. In-context Learning for Few-shot Multimodal Named Entity Recognition. In Findings of the Association for Computational Linguistics: EMNLP 2023. Association for Computational Linguistics, 2969-2979."},{"key":"e_1_3_2_1_5_1","volume-title":"Chain-of-thought prompt distillation for multimodal named entity and multimodal relation extraction. arXiv preprint arXiv:2306.14122","author":"Chen Feng","year":"2023","unstructured":"Feng Chen and Yujian Feng. 2023. Chain-of-thought prompt distillation for multimodal named entity and multimodal relation extraction. arXiv preprint arXiv:2306.14122 (2023)."},{"key":"e_1_3_2_1_6_1","volume-title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning. CoRR","author":"Chen Jun","year":"2023","unstructured":"Jun Chen, Deyao Zhu, Xiaoqian Shen, Xiang Li, Zechun Liu, Pengchuan Zhang, Raghuraman Krishnamoorthi, Vikas Chandra, Yunyang Xiong, and Mohamed Elhoseiny. 2023. MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning. CoRR, Vol. abs\/2310.09478 (2023)."},{"key":"e_1_3_2_1_7_1","first-page":"904","volume-title":"Hybrid Transformer with Multi-level Fusion for Multimodal Knowledge Graph Completion. In SIGIR '22: The 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","author":"Chen Xiang","year":"2022","unstructured":"Xiang Chen, Ningyu Zhang, Lei Li, Shumin Deng, Chuanqi Tan, Changliang Xu, Fei Huang, Luo Si, and Huajun Chen. 2022a. Hybrid Transformer with Multi-level Fusion for Multimodal Knowledge Graph Completion. In SIGIR '22: The 45th International ACM SIGIR Conference on Research and Development in Information Retrieval, Madrid, Spain, July 11 - 15, 2022, Enrique Amig\u00f3, Pablo Castells, Julio Gonzalo, Ben Carterette, J. Shane Culpepper, and Gabriella Kazai (Eds.). ACM, 904-915."},{"key":"e_1_3_2_1_8_1","first-page":"1607","volume-title":"Good Visual Guidance Make A Better Extractor: Hierarchical Visual Prefix for Multimodal Entity and Relation Extraction. In Findings of the Association for Computational Linguistics: NAACL","author":"Chen Xiang","year":"2022","unstructured":"Xiang Chen, Ningyu Zhang, Lei Li, Yunzhi Yao, Shumin Deng, Chuanqi Tan, Fei Huang, Luo Si, and Huajun Chen. 2022b. Good Visual Guidance Make A Better Extractor: Hierarchical Visual Prefix for Multimodal Entity and Relation Extraction. In Findings of the Association for Computational Linguistics: NAACL 2022. Association for Computational Linguistics, 1607-1618."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-031-37291-9","volume-title":"Knowledge Graphs Meet Multi-Modal Learning: A Comprehensive Survey. CoRR","volume":"2402","author":"Chen Zhuo","year":"2024","unstructured":"Zhuo Chen, Yichi Zhang, Yin Fang, Yuxia Geng, Lingbing Guo, Xiang Chen, Qian Li, Wen Zhang, Jiaoyan Chen, Yushan Zhu, Jiaqi Li, Xiaoze Liu, Jeff Z. Pan, Ningyu Zhang, and Huajun Chen. 2024. Knowledge Graphs Meet Multi-Modal Learning: A Comprehensive Survey. CoRR, Vol. abs\/2402.05391 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"A coefficient of agreement for nominal scales. Educational and psychological measurement","author":"Cohen Jacob","year":"1960","unstructured":"Jacob Cohen. 1960. A coefficient of agreement for nominal scales. Educational and psychological measurement, Vol. 20, 1 (1960), 37-46."},{"key":"e_1_3_2_1_11_1","first-page":"4171","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT 2019. Association for Computational Linguistics, 4171-4186."},{"key":"e_1_3_2_1_12_1","first-page":"8890","volume-title":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation, LREC\/COLING","author":"Ding Zepeng","year":"2024","unstructured":"Zepeng Ding, Wenhao Huang, Jiaqing Liang, Yanghua Xiao, and Deqing Yang. 2024a. Improving Recall of Large Language Models: A Model Collaboration Approach for Relational Triple Extraction. In Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation, LREC\/COLING 2024. ELRA and ICCL, 8890-8901."},{"key":"e_1_3_2_1_13_1","first-page":"8890","volume-title":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation, LREC\/COLING","author":"Ding Zepeng","year":"2024","unstructured":"Zepeng Ding, Wenhao Huang, Jiaqing Liang, Yanghua Xiao, and Deqing Yang. 2024b. Improving Recall of Large Language Models: A Model Collaboration Approach for Relational Triple Extraction. In Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation, LREC\/COLING 2024. ELRA and ICCL, 8890-8901."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the Fourth International Conference on Language Resources and Evaluation, LREC","author":"Doddington George R.","year":"2004","unstructured":"George R. Doddington, Alexis Mitchell, Mark A. Przybocki, Lance A. Ramshaw, Stephanie M. Strassel, and Ralph M. Weischedel. 2004. The Automatic Content Extraction (ACE) Program - Tasks, Data, and Evaluation. In Proceedings of the Fourth International Conference on Language Resources and Evaluation, LREC 2004. European Language Resources Association."},{"key":"e_1_3_2_1_15_1","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems","author":"Fei Hao","year":"2022","unstructured":"Hao Fei, Shengqiong Wu, Jingye Li, Bobo Li, Fei Li, Libo Qin, Meishan Zhang, Min Zhang, and Tat-Seng Chua. 2022. LasUIE: Unifying Information Extraction with Latent Adaptive Structure-aware Generative Language Model. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-3029"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612209"},{"key":"e_1_3_2_1_18_1","volume-title":"Multimodal Relational Triple Extraction with Query-based Entity Object Transformer. CoRR","author":"Hei Lei","year":"2024","unstructured":"Lei Hei, Ning An, Tingjing Liao, Qi Ma, Jiaqi Wang, and Feiliang Ren. 2024. Multimodal Relational Triple Extraction with Query-based Entity Object Transformer. CoRR, Vol. abs\/2408.08709 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations, ICLR","author":"Hu Edward J.","year":"2022","unstructured":"Edward J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations, ICLR 2022. OpenReview.net."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.613"},{"volume-title":"The Hungarian Method for the Assignment Problem. In 50 Years of Integer Programming 1958-2008 - From the Early Years to the State-of-the-Art","author":"Kuhn Harold W.","key":"e_1_3_2_1_21_1","unstructured":"Harold W. Kuhn. 2010. The Hungarian Method for the Assignment Problem. In 50 Years of Integer Programming 1958-2008 - From the Early Years to the State-of-the-Art. Springer, 29-47."},{"key":"e_1_3_2_1_22_1","first-page":"282","volume-title":"Proceedings of the Eighteenth International Conference on Machine Learning (ICML","author":"Lafferty John D.","year":"2001","unstructured":"John D. Lafferty, Andrew McCallum, and Fernando C. N. Pereira. 2001. Conditional Random Fields: Probabilistic Models for Segmenting and Labeling Sequence Data. In Proceedings of the Eighteenth International Conference on Machine Learning (ICML 2001). Morgan Kaufmann, 282-289."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.579"},{"key":"e_1_3_2_1_24_1","first-page":"2787","volume-title":"Prompting ChatGPT in MNER: Enhanced Multimodal Named Entity Recognition with Auxiliary Refined Knowledge. In Findings of the Association for Computational Linguistics: EMNLP","author":"Li Jinyuan","year":"2023","unstructured":"Jinyuan Li, Han Li, Zhuo Pan, Di Sun, Jiahao Wang, Wenkun Zhang, and Gang Pan. 2023a. Prompting ChatGPT in MNER: Enhanced Multimodal Named Entity Recognition with Auxiliary Refined Knowledge. In Findings of the Association for Computational Linguistics: EMNLP 2023. Association for Computational Linguistics, 2787-2802."},{"key":"e_1_3_2_1_25_1","first-page":"1302","volume-title":"ACL","author":"Li Jinyuan","year":"2024","unstructured":"Jinyuan Li, Han Li, Di Sun, Jiahao Wang, Wenkun Zhang, Zan Wang, and Gang Pan. 2024b. LLMs as Bridges: Reformulating Grounded Multimodal Named Entity Recognition. In Findings of the Association for Computational Linguistics, ACL 2024. Association for Computational Linguistics, 1302-1318."},{"key":"e_1_3_2_1_26_1","volume-title":"Advancing Grounded Multimodal Named Entity Recognition via LLM-Based Reformulation and Box-Based Segmentation. CoRR","author":"Li Jinyuan","year":"2024","unstructured":"Jinyuan Li, Ziyan Li, Han Li, Jianfei Yu, Rui Xia, Di Sun, and Gang Pan. 2024a. Advancing Grounded Multimodal Named Entity Recognition via LLM-Based Reformulation and Box-Based Segmentation. CoRR, Vol. abs\/2406.07268 (2024)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.806"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1129"},{"key":"e_1_3_2_1_29_1","first-page":"26286","volume-title":"Improved Baselines with Visual Instruction Tuning. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Yuheng Li, and Yong Jae Lee. 2024. Improved Baselines with Visual Instruction Tuning. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024. IEEE, 26286-26296."},{"key":"e_1_3_2_1_30_1","volume-title":"Visual Instruction Tuning. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual Instruction Tuning. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1236"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1185"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3445337"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.428"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.415"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1389"},{"key":"e_1_3_2_1_37_1","volume-title":"DSP: Discriminative Soft Prompts for Zero-Shot Entity and Relation Extraction. In Findings of the Association for Computational Linguistics: ACL","author":"Lv Bo","year":"2023","unstructured":"Bo Lv, Xin Liu, Shaojie Dai, Nayu Liu, Fan Yang, Ping Luo, and Yue Yu. 2023. DSP: Discriminative Soft Prompts for Zero-Shot Entity and Relation Extraction. In Findings of the Association for Computational Linguistics: ACL 2023. Association for Computational Linguistics."},{"key":"e_1_3_2_1_38_1","volume-title":"Deep and Interpretable Large Language Model Reasoning with Knowledge Graph-guided Retrieval. CoRR","author":"Ma Shengjie","year":"2024","unstructured":"Shengjie Ma, Chengjin Xu, Xuhui Jiang, Muzhi Li, Huaren Qu, and Jian Guo. 2024. Think-on-Graph 2.0: Deep and Interpretable Large Language Model Reasoning with Knowledge Graph-guided Retrieval. CoRR, Vol. abs\/2407.10805 (2024)."},{"key":"e_1_3_2_1_39_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. CoRR Vol. abs\/2303.08774 (2023)."},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT 2019","volume":"1","author":"Qin Kechen","year":"2019","unstructured":"Kechen Qin, Cheng Li, Virgil Pavlu, and Javed A. Aslam. 2019. Adapting RNN Sequence Prediction Model to Multi-label Set Prediction. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT 2019, Minneapolis, MN, USA, June 2-7, 2019, Volume 1 (Long and Short Papers), Jill Burstein, Christy Doran, and Thamar Solorio (Eds.). Association for Computational Linguistics, 3181-3190."},{"key":"e_1_3_2_1_41_1","unstructured":"Machel Reid Nikolay Savinov Denis Teplyashin Dmitry Lepikhin Timothy P. Lillicrap Jean-Baptiste Alayrac Radu Soricut Angeliki Lazaridou Orhan Firat Julian Schrittwieser Ioannis Antonoglou Rohan Anil Sebastian Borgeaud Andrew M. Dai Katie Millican Ethan Dyer Mia Glaese Thibault Sottiaux Benjamin Lee Fabio Viola Malcolm Reynolds Yuanzhong Xu James Molloy Jilin Chen Michael Isard Paul Barham Tom Hennigan Ross McIlroy Melvin Johnson Johan Schalkwyk Eli Collins Eliza Rutherford Erica Moreira Kareem Ayoub Megha Goel Clemens Meyer Gregory Thornton Zhen Yang Henryk Michalewski Zaheer Abbas Nathan Schucher Ankesh Anand Richard Ives James Keeling Karel Lenc Salem Haykal Siamak Shakeri Pranav Shyam Aakanksha Chowdhery Roman Ring Stephen Spencer Eren Sezener and et al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. CoRR Vol. abs\/2403.05530 (2024)."},{"key":"e_1_3_2_1_42_1","volume-title":"Multimodal Question Answering for Unified Information Extraction. CoRR","author":"Sun Yuxuan","year":"2023","unstructured":"Yuxuan Sun, Kai Zhang, and Yu Su. 2023. Multimodal Question Answering for Unified Information Extraction. CoRR, Vol. abs\/2310.03017 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Small Language Model Is a Good Guide for Large Language Model in Chinese Entity Relation Extraction. CoRR","author":"Tang Xuemei","year":"2024","unstructured":"Xuemei Tang, Jun Wang, and Qi Su. 2024. Small Language Model Is a Good Guide for Large Language Model in Chinese Entity Relation Extraction. CoRR, Vol. abs\/2402.14373 (2024)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i19.34302"},{"key":"e_1_3_2_1_45_1","first-page":"5998","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017. 5998-6008."},{"volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2023. Association for Computational Linguistics, 15566-15589","author":"Wadhwa Somin","key":"e_1_3_2_1_46_1","unstructured":"Somin Wadhwa, Silvio Amir, and Byron C. Wallace. 2023. Revisiting Relation Extraction in the era of Large Language Models. In Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2023. Association for Computational Linguistics, 15566-15589."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612322"},{"volume-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2023. 14734-14751","author":"Wu Shengqiong","key":"e_1_3_2_1_48_1","unstructured":"Shengqiong Wu, Hao Fei, Yixin Cao, Lidong Bing, and Tat-Seng Chua. [n.d.]. Information Screening whilst Exploiting! Multimodal Relation Extraction with Feature Denoising and Multimodal Topic Modeling. In Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2023. 14734-14751."},{"key":"e_1_3_2_1_49_1","volume-title":"MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities. CoRR","author":"Yu Weihao","year":"2024","unstructured":"Weihao Yu, Zhengyuan Yang, Linfeng Ren, Linjie Li, Jianfeng Wang, Kevin Lin, Chung-Ching Lin, Zicheng Liu, Lijuan Wang, and Xinchao Wang. 2024. MM-Vet v2: A Challenging Benchmark to Evaluate Large Multimodal Models for Integrated Capabilities. CoRR, Vol. abs\/2408.00765 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i9.26309"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3485107"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1203"},{"key":"e_1_3_2_1_53_1","first-page":"14498","volume-title":"ACL","author":"Zhang Meishan","year":"2024","unstructured":"Meishan Zhang, Hao Fei, Bin Wang, Shengqiong Wu, Yixin Cao, Fei Li, and Min Zhang. 2024. Recognizing Everything from All Modalities at Once: Grounded Multimodal Universal Information Extraction. In Findings of the Association for Computational Linguistics, ACL 2024. Association for Computational Linguistics, 14498-14511."},{"key":"e_1_3_2_1_54_1","first-page":"5579","volume-title":"VinVL: Revisiting Visual Representations in Vision-Language Models. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021","author":"Zhang Pengchuan","year":"2021","unstructured":"Pengchuan Zhang, Xiujun Li, Xiaowei Hu, Jianwei Yang, Lei Zhang, Lijuan Wang, Yejin Choi, and Jianfeng Gao. 2021. VinVL: Revisiting Visual Representations in Vision-Language Models. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, virtual, June 19-25, 2021. Computer Vision Foundation \/ IEEE, 5579-5588."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11962"},{"volume-title":"Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, EMNLP 2017. Association for Computational Linguistics, 35-45","author":"Zhang Yuhao","key":"e_1_3_2_1_56_1","unstructured":"Yuhao Zhang, Victor Zhong, Danqi Chen, Gabor Angeli, and Christopher D. Manning. 2017. Position-aware Attention and Supervised Data Improve Slot Filling. In Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, EMNLP 2017. Association for Computational Linguistics, 35-45."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.376"},{"key":"e_1_3_2_1_58_1","volume-title":"Multimodal Relation Extraction with Efficient Graph Alignment. In MM '21: ACM Multimedia Conference. ACM, 5298-5306","author":"Zheng Changmeng","year":"2021","unstructured":"Changmeng Zheng, Junhao Feng, Ze Fu, Yi Cai, Qing Li, and Tao Wang. 2021a. Multimodal Relation Extraction with Efficient Graph Alignment. In MM '21: ACM Multimedia Conference. ACM, 5298-5306."},{"key":"e_1_3_2_1_59_1","first-page":"1","volume-title":"MNRE: A Challenge Multimodal Dataset for Neural Relation Extraction with Visual Evidence in Social Media Posts. In 2021 IEEE International Conference on Multimedia and Expo, ICME","author":"Zheng Changmeng","year":"2021","unstructured":"Changmeng Zheng, Zhiwei Wu, Junhao Feng, Ze Fu, and Yi Cai. 2021b. MNRE: A Challenge Multimodal Dataset for Neural Relation Extraction with Visual Evidence in Social Media Posts. In 2021 IEEE International Conference on Multimedia and Expo, ICME 2021. IEEE, 1-6."},{"key":"e_1_3_2_1_60_1","volume-title":"Reefknot: A Comprehensive Benchmark for Relation Hallucination Evaluation, Analysis and Mitigation in Multimodal Large Language Models. CoRR","author":"Zheng Kening","year":"2024","unstructured":"Kening Zheng, Junkai Chen, Yibo Yan, Xin Zou, and Xuming Hu. 2024. Reefknot: A Comprehensive Benchmark for Relation Hallucination Evaluation, Analysis and Mitigation in Multimodal Large Language Models. CoRR, Vol. abs\/2408.09429 (2024)."},{"key":"e_1_3_2_1_61_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024. OpenReview.net."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755397","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T03:59:10Z","timestamp":1765339150000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755397"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":61,"alternative-id":["10.1145\/3746027.3755397","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755397","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}