{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:16:50Z","timestamp":1781587010299,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612470","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"3209-3218","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["MCG-MNER: A Multi-Granularity Cross-Modality Generative Framework for Multimodal NER with Instruction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4178-8952","authenticated-orcid":false,"given":"Junjie","family":"Wu","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Soochow University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2570-6969","authenticated-orcid":false,"given":"Chen","family":"Gong","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, School of Computer Science and Technology, Soochow University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1077-9033","authenticated-orcid":false,"given":"Ziqiang","family":"Cao","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, School of Computer Science and Technology, Soochow University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6882-6181","authenticated-orcid":false,"given":"Guohong","family":"Fu","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, School of Computer Science and Technology, Soochow University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00061"},{"key":"e_1_3_2_2_2_1","volume-title":"BEIT: BERT pre-training of image transformers. arXiv preprint arXiv:2106.08254","author":"Bao Hangbo","year":"2021","unstructured":"Hangbo Bao, Li Dong, Songhao Piao, and Furu Wei. 2021. BEIT: BERT pre-training of image transformers. arXiv preprint arXiv:2106.08254 (2021)."},{"key":"e_1_3_2_2_3_1","volume-title":"Neural Information Processing Systems (2020)","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al. 2020. Language models are few-shot learners. Neural Information Processing Systems (2020), 1877--1901."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1171"},{"key":"e_1_3_2_2_5_1","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics. 2374--2387","author":"Chen Xiang","year":"2022","unstructured":"Xiang Chen, Lei Li, Shumin Deng, Chuanqi Tan, Changliang Xu, Fei Huang, Luo Si, Huajun Chen, and Ningyu Zhang. 2022a. LightNER: A lightweight tuning paradigm for low-resource NER via pluggable prompting. In Proceedings of the 29th International Conference on Computational Linguistics. 2374--2387."},{"key":"e_1_3_2_2_6_1","volume-title":"Good visual guidance makes a better extractor: Hierarchical visual prefix for multimodal entity and relation extraction. arXiv preprint arXiv:2205.03521","author":"Chen Xiang","year":"2022","unstructured":"Xiang Chen, Ningyu Zhang, Lei Li, Yunzhi Yao, Shumin Deng, Chuanqi Tan, Fei Huang, Luo Si, and Huajun Chen. 2022b. Good visual guidance makes a better extractor: Hierarchical visual prefix for multimodal entity and relation extraction. arXiv preprint arXiv:2205.03521 (2022)."},{"key":"e_1_3_2_2_7_1","volume-title":"Enhancing multimodal entity and relation extraction with variational information bottleneck. arXiv preprint arXiv:2304.02328","author":"Cui Shiyao","year":"2023","unstructured":"Shiyao Cui, Jiangxia Cao, Xin Cong, Jiawei Sheng, Quangang Li, Tingwen Liu, and Jinqiao Shi. 2023. Enhancing multimodal entity and relation extraction with variational information bottleneck. arXiv preprint arXiv:2304.02328 (2023)."},{"key":"e_1_3_2_2_8_1","volume-title":"International Conference on Machine Learning. 7865--7885","author":"Giovanni Francesco Di","year":"2023","unstructured":"Francesco Di Giovanni, Lorenzo Giusti, Federico Barbero, Giulia Luise, Pietro Lio, and Michael M Bronstein. 2023. On over-squashing in message passing neural networks: The impact of width, depth, and topology. In International Conference on Machine Learning. 7865--7885."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26504"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW58289.2023.00042"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40747-022-00714-9"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548427"},{"key":"e_1_3_2_2_14_1","volume-title":"Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 4171--4186","author":"Ming-Wei Chang Jacob Devlin","year":"2019","unstructured":"Jacob Devlin Ming-Wei Chang Kenton and Lee Kristina Toutanova. 2019. BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 4171--4186."},{"key":"e_1_3_2_2_15_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.364"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.144"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1185"},{"key":"e_1_3_2_2_21_1","volume-title":"Advances in Neural Information Processing Systems","volume":"32","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. ViLBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in Neural Information Processing Systems, Vol. 32 (2019)."},{"key":"e_1_3_2_2_22_1","volume-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics. 3470--3487","author":"Mishra Swaroop","year":"2022","unstructured":"Swaroop Mishra, Daniel Khashabi, Chitta Baral, and Hannaneh Hajishirzi. 2022. Cross-task generalization via natural language crowdsourcing Iinstructions. In Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics. 3470--3487."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1078"},{"key":"e_1_3_2_2_24_1","volume-title":"GPT-4 technical report. arXiv","author":"R","year":"2023","unstructured":"R OpenAI. 2023. GPT-4 technical report. arXiv (2023), 2303--08774."},{"key":"e_1_3_2_2_25_1","unstructured":"Adam Paszke Sam Gross Francisco Massa Adam Lerer James Bradbury Gregory Chanan Trevor Killeen Zeming Lin Natalia Gimelshein Luca Antiga et al. 2019. PyTorch: An imperative style high-performance deep learning library. Neural Information Processing Systems (2019) 8026--8037."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.217"},{"key":"e_1_3_2_2_27_1","volume-title":"International conference on machine learning. 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. 8748--8763."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.5555\/3455716.3455856"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.20"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i15.17633"},{"key":"e_1_3_2_2_32_1","volume-title":"Attention is all you need. Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Neural Information Processing Systems (2017), 6000--6010."},{"key":"e_1_3_2_2_33_1","volume-title":"Proceedings of the Ninth Conference of the European Chapter of the Association for Computational Linguistics. 173--179","author":"Veenstra J","year":"1999","unstructured":"J Veenstra and EF Tjong Kim Sang. 1999. Representing text chunks. In Proceedings of the Ninth Conference of the European Chapter of the Association for Computational Linguistics. 173--179."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1272"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3221017"},{"key":"e_1_3_2_2_36_1","volume-title":"Instructionner: A multi-task instruction-based generative framework for few-shot ner. arXiv preprint arXiv:2203.03903","author":"Wang Liwen","year":"2022","unstructured":"Liwen Wang, Rumei Li, Yang Yan, Yuanmeng Yan, Sirui Wang, Wei Wu, and Weiran Xu. 2022c. Instructionner: A multi-task instruction-based generative framework for few-shot ner. arXiv preprint arXiv:2203.03903 (2022)."},{"key":"e_1_3_2_2_37_1","first-page":"545","article-title":"Multimodal named entity recognition with bottleneck fusion and contrastive learning","volume":"106","author":"Peng WANG, Xiaohang CHEN, Ziyu","year":"2023","unstructured":"Peng WANG, Xiaohang CHEN, Ziyu SHANG, and Wenjun KE. 2023. Multimodal named entity recognition with bottleneck fusion and contrastive learning. IEICE Transactions on Information and Systems, Vol. 106, 4 (2023), 545--555.","journal-title":"IEICE Transactions on Information and Systems"},{"key":"e_1_3_2_2_38_1","volume-title":"International Conference on Machine Learning. 23318--23340","author":"Wang Peng","year":"2022","unstructured":"Peng Wang, An Yang, Rui Men, Junyang Lin, Shuai Bai, Zhikang Li, Jianxin Ma, Chang Zhou, Jingren Zhou, and Hongxia Yang. 2022 f. OFA: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In International Conference on Machine Learning. 23318--23340."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-emnlp.437"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.232"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.328"},{"key":"e_1_3_2_2_43_1","volume-title":"Learning combinatorial prompts for universal controllable image captioning. arXiv preprint arXiv:2303.06338","author":"Wang Zhen","year":"2023","unstructured":"Zhen Wang, Jun Xiao, Lei Chen, Fei Gao, Jian Shao, and Long Chen. 2023. Learning combinatorial prompts for universal controllable image captioning. arXiv preprint arXiv:2303.06338 (2023)."},{"key":"e_1_3_2_2_44_1","volume-title":"Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le.","author":"Wei Jason","year":"2021","unstructured":"Jason Wei, Maarten Bosma, Vincent Y Zhao, Kelvin Guu, Adams Wei Yu, Brian Lester, Nan Du, Andrew M Dai, and Quoc V Le. 2021. Finetuned language models are zero-shot learners. arXiv preprint arXiv:2109.01652 (2021)."},{"key":"e_1_3_2_2_45_1","volume-title":"Towards a unified model for generating answers and explanations in visual question answering. arXiv preprint arXiv:2301.10799","author":"Whitehouse Chenxi","year":"2023","unstructured":"Chenxi Whitehouse, Tillman Weyde, and Pranava Madhyastha. 2023. Towards a unified model for generating answers and explanations in visual question answering. arXiv preprint arXiv:2301.10799 (2023)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-60450-9_12"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413650"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498475"},{"key":"e_1_3_2_2_49_1","volume-title":"Multimodal learning With transformers: A survey","author":"Xu Peng","year":"2023","unstructured":"Peng Xu, Xiatian Zhu, and David A Clifton. 2023. Multimodal learning With transformers: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023), 1--20."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00478"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.306"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17687"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11962"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548228"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME51207.2021.9428240"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3476968"},{"key":"e_1_3_2_2_57_1","volume-title":"2021 IEEE International Conference on Multimedia and Expo. 1--6.","author":"Zheng Changmeng","year":"2021","unstructured":"Changmeng Zheng, Zhiwei Wu, Junhao Feng, Ze Fu, and Yi Cai. 2021b. MNER: A challenge multimodal dataset for neural relation extraction with visual evidence in social media posts. In 2021 IEEE International Conference on Multimedia and Expo. 1--6."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3013398"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.7005"},{"key":"e_1_3_2_2_60_1","volume-title":"2021 IEEE\/CVF International Conference on Computer Vision. 12757--12766","author":"Zhou Zhongkai","year":"2021","unstructured":"Zhongkai Zhou, Xinnan Fan, Pengfei Shi, and Yuanxue Xin. 2021. R-msfn: Recurrent multi-scale feature modulation for monocular depth estimating. In 2021 IEEE\/CVF International Conference on Computer Vision. 12757--12766."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612470","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612470","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:57:26Z","timestamp":1755820646000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612470"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":60,"alternative-id":["10.1145\/3581783.3612470","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612470","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}