{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T14:39:30Z","timestamp":1777127970492,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["No.2021ZD0113000"],"award-info":[{"award-number":["No.2021ZD0113000"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3548387","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:43:12Z","timestamp":1665416592000},"page":"5274-5282","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["AI-VQA"],"prefix":"10.1145","author":[{"given":"Rengang","family":"Li","sequence":"first","affiliation":[{"name":"Inspur (Beijing) Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cong","family":"Xu","sequence":"additional","affiliation":[{"name":"Inspur (Beijing) Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenhua","family":"Guo","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baoyu","family":"Fan","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Runze","family":"Zhang","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Liu","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yaqian","family":"Zhao","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weifeng","family":"Gong","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Endong","family":"Wang","sequence":"additional","affiliation":[{"name":"Inspur Electronic Information Industry Co., Ltd. &amp; State Key Laboratory of High-end Server &amp; Storage Technology, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_2_3_1","volume-title":"Zero-Shot Visual Question Answering Using Knowledge Graph. In International Semantic Web Conference. Springer, 146--162","author":"Chen Zhuo","year":"2021","unstructured":"Zhuo Chen , Jiaoyan Chen , Yuxia Geng , Jeff Z Pan , Zonggang Yuan , and Huajun Chen . 2021 . Zero-Shot Visual Question Answering Using Knowledge Graph. In International Semantic Web Conference. Springer, 146--162 . Zhuo Chen, Jiaoyan Chen, Yuxia Geng, Jeff Z Pan, Zonggang Yuan, and Huajun Chen. 2021. Zero-Shot Visual Question Answering Using Knowledge Graph. In International Semantic Web Conference. Springer, 146--162."},{"key":"e_1_3_2_2_4_1","volume-title":"Contextual Multi-Scale Feature Learning for Person Re-Identification. In MM '20: The 28th ACM International Conference on Multimedia, Virtual Event \/ Seattle, WA, USA","author":"Fan Baoyu","year":"2020","unstructured":"Baoyu Fan , Li Wang , Runze Zhang , Zhenhua Guo , Yaqian Zhao , Rengang Li , and Weifeng Gong . 2020 . Contextual Multi-Scale Feature Learning for Person Re-Identification. In MM '20: The 28th ACM International Conference on Multimedia, Virtual Event \/ Seattle, WA, USA , October 12-16, 2020, Chang Wen Chen, Rita Cucchiara, Xian-Sheng Hua, Guo-Jun Qi, Elisa Ricci, Zhengyou Zhang, and Roger Zimmermann (Eds.). ACM, 655--663. https:\/\/doi.org\/10.1145\/3394171.3414038 Baoyu Fan, Li Wang, Runze Zhang, Zhenhua Guo, Yaqian Zhao, Rengang Li, and Weifeng Gong. 2020. Contextual Multi-Scale Feature Learning for Person Re-Identification. In MM '20: The 28th ACM International Conference on Multimedia, Virtual Event \/ Seattle, WA, USA, October 12-16, 2020, Chang Wen Chen, Rita Cucchiara, Xian-Sheng Hua, Guo-Jun Qi, Elisa Ricci, Zhengyou Zhang, and Roger Zimmermann (Eds.). ACM, 655--663. https:\/\/doi.org\/10.1145\/3394171.3414038"},{"key":"e_1_3_2_2_5_1","volume-title":"Daylen Yang, Anna Rohrbach, Trevor Darrell, and Marcus Rohrbach.","author":"Fukui Akira","year":"2016","unstructured":"Akira Fukui , Dong Huk Park , Daylen Yang, Anna Rohrbach, Trevor Darrell, and Marcus Rohrbach. 2016 . Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding. In EMNLP. Akira Fukui, Dong Huk Park, Daylen Yang, Anna Rohrbach, Trevor Darrell, and Marcus Rohrbach. 2016. Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding. In EMNLP."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1422953112"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00686"},{"key":"e_1_3_2_2_9_1","volume-title":"GQA: A New Dataset for Real- World Visual Reasoning and Compositional Question Answering. Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Hudson Drew A","year":"2019","unstructured":"Drew A Hudson and Christopher D Manning . 2019 . GQA: A New Dataset for Real- World Visual Reasoning and Compositional Question Answering. Conference on Computer Vision and Pattern Recognition (CVPR) (2019). Drew A Hudson and Christopher D Manning. 2019. GQA: A New Dataset for Real- World Visual Reasoning and Compositional Question Answering. Conference on Computer Vision and Pattern Recognition (CVPR) (2019)."},{"key":"e_1_3_2_2_10_1","volume-title":"Jeff Da, Keisuke Sakaguchi, Antoine Bosselut, and Yejin Choi.","author":"Hwang Jena D","year":"2020","unstructured":"Jena D Hwang , Chandra Bhagavatula , Ronan Le Bras , Jeff Da, Keisuke Sakaguchi, Antoine Bosselut, and Yejin Choi. 2020 . Comet-atomic 2020: On symbolic and neural commonsense knowledge graphs. arXiv preprint arXiv:2010.05953 (2020). Jena D Hwang, Chandra Bhagavatula, Ronan Le Bras, Jeff Da, Keisuke Sakaguchi, Antoine Bosselut, and Yejin Choi. 2020. Comet-atomic 2020: On symbolic and neural commonsense knowledge graphs. arXiv preprint arXiv:2010.05953 (2020)."},{"key":"e_1_3_2_2_11_1","volume-title":"Substitute, Search: A New Benchmark for Knowledge-Augmented Visual Question Answering. arXiv preprint arXiv:2103.05568","author":"Jain Aman","year":"2021","unstructured":"Aman Jain , Mayank Kothyari , Vishwajeet Kumar , Preethi Jyothi , Ganesh Ramakrishnan , and Soumen Chakrabarti . 2021. Select , Substitute, Search: A New Benchmark for Knowledge-Augmented Visual Question Answering. arXiv preprint arXiv:2103.05568 ( 2021 ). Aman Jain, Mayank Kothyari, Vishwajeet Kumar, Preethi Jyothi, Ganesh Ramakrishnan, and Soumen Chakrabarti. 2021. Select, Substitute, Search: A New Benchmark for Knowledge-Augmented Visual Question Answering. arXiv preprint arXiv:2103.05568 (2021)."},{"key":"e_1_3_2_2_12_1","unstructured":"D. Jia D. Wei R. Socher L. J. Li L. Kai and F. F. Li. 2009. ImageNet: A large-scale hierarchical image database. 248--255.  D. Jia D. Wei R. Socher L. J. Li L. Kai and F. F. Li. 2009. ImageNet: A large-scale hierarchical image database. 248--255."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.215"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00128"},{"key":"e_1_3_2_2_15_1","first-page":"1571","article-title":"Bilinear attention networks","volume":"31","author":"Kim Jin-Hwa","year":"2018","unstructured":"Jin-Hwa Kim , Jaehyun Jun , and Byoung-Tak Zhang . 2018 . Bilinear attention networks . Advances in Neural Information Processing Systems 31 (2018), 1571 -- 1581 . Jin-Hwa Kim, Jaehyun Jun, and Byoung-Tak Zhang. 2018. Bilinear attention networks. Advances in Neural Information Processing Systems 31 (2018), 1571--1581.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Yuke Zhu Oliver Groth Justin Johnson Kenji Hata Joshua Kravitz Stephanie Chen Yannis Kalantidis Li-Jia Li David A Shamma etal 2017. Visual genome: Connecting language and vision using crowdsourced dense image annotations. International journal of computer vision 123 1 (2017) 32--73.  Ranjay Krishna Yuke Zhu Oliver Groth Justin Johnson Kenji Hata Joshua Kravitz Stephanie Chen Yannis Kalantidis Li-Jia Li David A Shamma et al. 2017. Visual genome: Connecting language and vision using crowdsourced dense image annotations. International journal of computer vision 123 1 (2017) 32--73.","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3124317"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3161076"},{"key":"e_1_3_2_2_19_1","volume-title":"Joint active learning with feature selection via cur matrix decomposition","author":"Li Changsheng","year":"2018","unstructured":"Changsheng Li , Xiangfeng Wang , Weishan Dong , Junchi Yan , Qingshan Liu , and Hongyuan Zha . 2018. Joint active learning with feature selection via cur matrix decomposition . IEEE transactions on pattern analysis and machine intelligence 41, 6 ( 2018 ), 1382--1396. Changsheng Li, Xiangfeng Wang, Weishan Dong, Junchi Yan, Qingshan Liu, and Hongyuan Zha. 2018. Joint active learning with feature selection via cur matrix decomposition. IEEE transactions on pattern analysis and machine intelligence 41, 6 (2018), 1382--1396."},{"key":"e_1_3_2_2_20_1","volume-title":"Dynamic structure embedded online multiple-output regression for streaming data","author":"Li Changsheng","year":"2018","unstructured":"Changsheng Li , Fan Wei , Weishan Dong , Xiangfeng Wang , Qingshan Liu , and Xin Zhang . 2018. Dynamic structure embedded online multiple-output regression for streaming data . IEEE transactions on pattern analysis and machine intelligence 41, 2 ( 2018 ), 323--336. Changsheng Li, Fan Wei, Weishan Dong, Xiangfeng Wang, Qingshan Liu, and Xin Zhang. 2018. Dynamic structure embedded online multiple-output regression for streaming data. IEEE transactions on pattern analysis and machine intelligence 41, 2 (2018), 323--336."},{"key":"e_1_3_2_2_21_1","volume-title":"Tell-and-answer: Towards explainable visual question answering using attributes and captions. arXiv preprint arXiv:1801.09041","author":"Li Qing","year":"2018","unstructured":"Qing Li , Jianlong Fu , Dongfei Yu , Tao Mei , and Jiebo Luo . 2018 . Tell-and-answer: Towards explainable visual question answering using attributes and captions. arXiv preprint arXiv:1801.09041 (2018). Qing Li, Jianlong Fu, Dongfei Yu, Tao Mei, and Jiebo Luo. 2018. Tell-and-answer: Towards explainable visual question answering using attributes and captions. arXiv preprint arXiv:1801.09041 (2018)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"e_1_3_2_2_23_1","volume-title":"ConceptNet-a practical commonsense reasoning tool-kit. BT technology journal 22, 4","author":"Liu Hugo","year":"2004","unstructured":"Hugo Liu and Push Singh . 2004. ConceptNet-a practical commonsense reasoning tool-kit. BT technology journal 22, 4 ( 2004 ), 211--226. Hugo Liu and Push Singh. 2004. ConceptNet-a practical commonsense reasoning tool-kit. BT technology journal 22, 4 (2004), 211--226."},{"key":"e_1_3_2_2_24_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu , Myle Ott , Naman Goyal , Jingfei Du , Mandar Joshi , Danqi Chen , Omer Levy , Mike Lewis , Luke Zettlemoyer , and Veselin Stoyanov . 2019 . Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019). Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_2_25_1","volume-title":"A multi-world approach to question answering about real-world scenes based on uncertain input. Advances in neural information processing systems 27","author":"Malinowski Mateusz","year":"2014","unstructured":"Mateusz Malinowski and Mario Fritz . 2014. A multi-world approach to question answering about real-world scenes based on uncertain input. Advances in neural information processing systems 27 ( 2014 ), 1682--1690. Mateusz Malinowski and Mario Fritz. 2014. A multi-world approach to question answering about real-world scenes based on uncertain input. Advances in neural information processing systems 27 (2014), 1682--1690."},{"key":"e_1_3_2_2_26_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 3195--3204","author":"Marino Kenneth","year":"2019","unstructured":"Kenneth Marino , Mohammad Rastegari , Ali Farhadi , and Roozbeh Mottaghi . 2019 . Ok-vqa:Avisual question answering benchmark requiring external knowledge . In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 3195--3204 . Kenneth Marino, Mohammad Rastegari, Ali Farhadi, and Roozbeh Mottaghi. 2019. Ok-vqa:Avisual question answering benchmark requiring external knowledge. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 3195--3204."},{"key":"e_1_3_2_2_27_1","volume-title":"Out of the box: Reasoning with graph convolution nets for factual visual question answering. arXiv preprint arXiv:1811.00538","author":"Narasimhan Medhini","year":"2018","unstructured":"Medhini Narasimhan , Svetlana Lazebnik , and Alexander G Schwing . 2018. Out of the box: Reasoning with graph convolution nets for factual visual question answering. arXiv preprint arXiv:1811.00538 ( 2018 ). Medhini Narasimhan, Svetlana Lazebnik, and Alexander G Schwing. 2018. Out of the box: Reasoning with graph convolution nets for factual visual question answering. arXiv preprint arXiv:1811.00538 (2018)."},{"key":"e_1_3_2_2_28_1","volume-title":"Imagebert: Cross-modal pre-training with large-scale weak-supervised image text data. arXiv preprint arXiv:2001.07966","author":"Qi Di","year":"2020","unstructured":"Di Qi , Lin Su , Jia Song , Edward Cui , Taroon Bharti , and Arun Sacheti . 2020 . Imagebert: Cross-modal pre-training with large-scale weak-supervised image text data. arXiv preprint arXiv:2001.07966 (2020). Di Qi, Lin Su, Jia Song, Edward Cui, Taroon Bharti, and Arun Sacheti. 2020. Imagebert: Cross-modal pre-training with large-scale weak-supervised image text data. arXiv preprint arXiv:2001.07966 (2020)."},{"key":"e_1_3_2_2_29_1","volume-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)","author":"Reimers Nils","year":"1865","unstructured":"Nils Reimers and Iryna Gurevych . 2019. Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks . In Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP) . Association for Computational Linguistics , Hong Kong , China, 3982--3992. https:\/\/doi.org\/10. 1865 3\/v1\/D19-1410 Nils Reimers and Iryna Gurevych. 2019. Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. In Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP). Association for Computational Linguistics, Hong Kong, China, 3982--3992. https:\/\/doi.org\/10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_2_30_1","volume-title":"Knowledge-Supervised Learning: Knowledge Consensus Constraints for Person Re-Identification. In MM '21: ACM Multimedia Conference","author":"Wang Li","year":"2021","unstructured":"Li Wang , Baoyu Fan , Zhenhua Guo , Yaqian Zhao , Runze Zhang , Rengang Li , Weifeng Gong , and Endong Wang . 2021 . Knowledge-Supervised Learning: Knowledge Consensus Constraints for Person Re-Identification. In MM '21: ACM Multimedia Conference , Virtual Event, China, October 20 - 24 , 2021, Heng Tao Shen, Yueting Zhuang, John R. Smith, Yang Yang, Pablo Cesar, Florian Metze, and Balakrishnan Prabhakaran (Eds.). ACM, 1866--1874. https: \/\/doi.org\/10.1145\/3474085.3475340 Li Wang, Baoyu Fan, Zhenhua Guo, Yaqian Zhao, Runze Zhang, Rengang Li, Weifeng Gong, and Endong Wang. 2021. Knowledge-Supervised Learning: Knowledge Consensus Constraints for Person Re-Identification. In MM '21: ACM Multimedia Conference, Virtual Event, China, October 20 - 24, 2021, Heng Tao Shen, Yueting Zhuang, John R. Smith, Yang Yang, Pablo Cesar, Florian Metze, and Balakrishnan Prabhakaran (Eds.). ACM, 1866--1874. https: \/\/doi.org\/10.1145\/3474085.3475340"},{"key":"e_1_3_2_2_31_1","volume-title":"Fvqa: Fact-based visual question answering","author":"Wang Peng","year":"2017","unstructured":"Peng Wang , Qi Wu , Chunhua Shen , Anthony Dick , and Anton Van Den Hengel . 2017 . Fvqa: Fact-based visual question answering . IEEE transactions on pattern analysis and machine intelligence 40, 10 (2017), 2413--2427. Peng Wang, Qi Wu, Chunhua Shen, Anthony Dick, and Anton Van Den Hengel. 2017. Fvqa: Fact-based visual question answering. IEEE transactions on pattern analysis and machine intelligence 40, 10 (2017), 2413--2427."},{"key":"e_1_3_2_2_32_1","volume-title":"Explicit knowledge-based reasoning for visual question answering. arXiv preprint arXiv:1511.02570","author":"Wang Peng","year":"2015","unstructured":"Peng Wang , Qi Wu , Chunhua Shen , Anton van den Hengel , and Anthony Dick . 2015. Explicit knowledge-based reasoning for visual question answering. arXiv preprint arXiv:1511.02570 ( 2015 ). Peng Wang, Qi Wu, Chunhua Shen, Anton van den Hengel, and Anthony Dick. 2015. Explicit knowledge-based reasoning for visual question answering. arXiv preprint arXiv:1511.02570 (2015)."},{"key":"e_1_3_2_2_33_1","volume-title":"Towards Reasoning Ability in Scene Text Visual Question Answering","author":"Wang Qingqing","unstructured":"Qingqing Wang , Liqiang Xiao , Yue Lu , Yaohui Jin , and Hao He. 2021. Towards Reasoning Ability in Scene Text Visual Question Answering . Association for Computing Machinery , New York, NY, USA , 2281--2289. https:\/\/doi.org\/10.1145\/ 3474085.3475390 Qingqing Wang, Liqiang Xiao, Yue Lu, Yaohui Jin, and Hao He. 2021. Towards Reasoning Ability in Scene Text Visual Question Answering. Association for Computing Machinery, New York, NY, USA, 2281--2289. https:\/\/doi.org\/10.1145\/ 3474085.3475390"},{"key":"e_1_3_2_2_34_1","volume-title":"Deconfounded and Explainable Interactive Vision-Language Retrieval of Complex Scenes","author":"Yu Tong","unstructured":"JundaWu, Tong Yu , and Shuai Li. 2021. Deconfounded and Explainable Interactive Vision-Language Retrieval of Complex Scenes . Association for Computing Machinery , New York, NY, USA , 2103--2111. https:\/\/doi.org\/10.1145\/3474085.3475366 JundaWu, Tong Yu, and Shuai Li. 2021. Deconfounded and Explainable Interactive Vision-Language Retrieval of Complex Scenes. Association for Computing Machinery, New York, NY, USA, 2103--2111. https:\/\/doi.org\/10.1145\/3474085.3475366"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.202"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00688"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3104166"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00553"}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548387","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3548387","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:00:44Z","timestamp":1750186844000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548387"}},"subtitle":["Visual Question Answering based on Agent Interaction with Interpretability"],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":38,"alternative-id":["10.1145\/3503161.3548387","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3548387","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}