{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T11:16:17Z","timestamp":1783422977505,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":81,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.62072147"],"award-info":[{"award-number":["No.62072147"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["No.LR22F020001 and No.LDT23F02025F02"],"award-info":[{"award-number":["No.LR22F020001 and No.LDT23F02025F02"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3611939","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"387-396","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Parameter-Efficient Transfer Learning for Audio-Visual-Language Tasks"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2558-7939","authenticated-orcid":false,"given":"Hongye","family":"Liu","sequence":"first","affiliation":[{"name":"School of Mechanical and Electrical Engineering, China JiLiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3996-1590","authenticated-orcid":false,"given":"Xianhai","family":"Xie","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6717-0373","authenticated-orcid":false,"given":"Yang","family":"Gao","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8407-1137","authenticated-orcid":false,"given":"Zhou","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. NIPS","author":"Akbari Hassan","year":"2021","unstructured":"Hassan Akbari, Liangzhe Yuan, Rui Qian, Wei-Hong Chuang, Shih-Fu Chang, Yin Cui, and Boqing Gong. 2021. Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. NIPS (2021)."},{"key":"e_1_3_2_1_2_1","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katie Millican Malcolm Reynolds et al. 2022. Flamingo: a visual language model for few-shot learning. arXiv preprint arXiv:2204.14198 (2022)."},{"key":"e_1_3_2_1_3_1","volume-title":"Vqa: Visual question answering. In ICCV.","author":"Antol Stanislaw","year":"2015","unstructured":"Stanislaw Antol, Aishwarya Agrawal, Jiasen Lu, Margaret Mitchell, Dhruv Batra, C Lawrence Zitnick, and Devi Parikh. 2015. Vqa: Visual question answering. In ICCV."},{"key":"e_1_3_2_1_4_1","volume-title":"Kriti Aggarwal, Subhojit Som, and Furu Wei.","author":"Bao Hangbo","year":"2021","unstructured":"Hangbo Bao, Wenhui Wang, Li Dong, Qiang Liu, Owais Khan Mohammed, Kriti Aggarwal, Subhojit Som, and Furu Wei. 2021. Vlmo: Unified vision-language pre-training with mixture-of-modality-experts. arXiv preprint arXiv:2111.02358 (2021)."},{"key":"e_1_3_2_1_5_1","volume-title":"Conv-Adapter: Exploring Parameter Efficient Transfer Learning for ConvNets. arXiv preprint arXiv:2208.07463","author":"Chen Hao","year":"2022","unstructured":"Hao Chen, Ran Tao, Han Zhang, Yidong Wang, Wei Ye, Jindong Wang, Guosheng Hu, and Marios Savvides. 2022. Conv-Adapter: Exploring Parameter Efficient Transfer Learning for ConvNets. arXiv preprint arXiv:2208.07463 (2022)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Junwen Chen and Yu Kong. 2021. Explainable video entailment with grounded visual evidence. In CVPR.","DOI":"10.1109\/ICCV48922.2021.00203"},{"key":"e_1_3_2_1_7_1","volume-title":"Faisal Ahmed, Zhe Gan, Yu Cheng, and Jingjing Liu.","author":"Chen Yen-Chun","year":"2020","unstructured":"Yen-Chun Chen, Linjie Li, Licheng Yu, Ahmed El Kholy, Faisal Ahmed, Zhe Gan, Yu Cheng, and Jingjing Liu. 2020. Uniter: Universal image-text representation learning. In ECCV."},{"key":"e_1_3_2_1_8_1","volume-title":"Vision transformer adapter for dense predictions. arXiv preprint arXiv:2205.08534","author":"Chen Zhe","year":"2022","unstructured":"Zhe Chen, Yuchen Duan, Wenhai Wang, Junjun He, Tong Lu, Jifeng Dai, and Yu Qiao. 2022. Vision transformer adapter for dense predictions. arXiv preprint arXiv:2205.08534 (2022)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Zhao-Min Chen Xiu-Shen Wei Peng Wang and Yanwen Guo. 2019. Multi-label image recognition with graph convolutional networks. In CVPR.","DOI":"10.1109\/CVPR.2019.00532"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475251"},{"key":"e_1_3_2_1_11_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Zi-Yi Dou Yichong Xu Zhe Gan Jianfeng Wang Shuohang Wang Lijuan Wang Chenguang Zhu Pengchuan Zhang Lu Yuan Nanyun Peng et al. 2022. An empirical study of training end-to-end vision-and-language transformers. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01763"},{"key":"e_1_3_2_1_13_1","unstructured":"Chenyou Fan Xiaofan Zhang Shu Zhang Wensheng Wang Chi Zhang and Heng Huang. 2019. Heterogeneous memory enhanced multimodal attention model for video question answering. In CVPR."},{"key":"e_1_3_2_1_14_1","volume-title":"Temporal reasoning via audio question answering. TASLP","author":"Fayek Haytham M","year":"2020","unstructured":"Haytham M Fayek and Justin Johnson. 2020. Temporal reasoning via audio question answering. TASLP (2020)."},{"key":"e_1_3_2_1_15_1","volume-title":"Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, et al. 2020. Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)."},{"key":"e_1_3_2_1_16_1","volume-title":"Audioclip: Extending clip to image, text and audio","author":"Guzhov Andrey","year":"2022","unstructured":"Andrey Guzhov, Federico Raue, J\u00f6rn Hees, and Andreas Dengel. 2022. Audioclip: Extending clip to image, text and audio. In ICASSP. IEEE."},{"key":"e_1_3_2_1_17_1","volume-title":"Louis-Philippe Morency, et al.","author":"Hasan Md Kamrul","year":"2019","unstructured":"Md Kamrul Hasan, Wasifur Rahman, Amir Zadeh, Jianyuan Zhong, Md Iftekhar Tanveer, Louis-Philippe Morency, et al. 2019. UR-FUNNY: A multimodal language dataset for understanding humor. arXiv preprint arXiv:1904.06618 (2019)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413678"},{"key":"e_1_3_2_1_19_1","volume-title":"Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly.","author":"Houlsby Neil","year":"2019","unstructured":"Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin De Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly. 2019. Parameter-efficient transfer learning for NLP. In ICML."},{"key":"e_1_3_2_1_20_1","volume-title":"International conference on machine learning. PMLR, 4651--4664","author":"Jaegle Andrew","year":"2021","unstructured":"Andrew Jaegle, Felix Gimeno, Andy Brock, Oriol Vinyals, Andrew Zisserman, and Joao Carreira. 2021. Perceiver: General perception with iterative attention. In International conference on machine learning. PMLR, 4651--4664."},{"key":"e_1_3_2_1_21_1","unstructured":"Chen Ju Tengda Han Kunhao Zheng Ya Zhang and Weidi Xie. 2022. Prompting visual-language models for efficient video understanding. In ECCV."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of naacL-HLT","volume":"1","author":"Ming-Wei Chang Jacob Devlin","year":"2019","unstructured":"Jacob Devlin Ming-Wei Chang Kenton and Lee Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of naacL-HLT, Vol. 1. 2."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1611835114"},{"key":"e_1_3_2_1_24_1","unstructured":"Thao Minh Le Vuong Le Svetha Venkatesh and Truyen Tran. 2020. Hierarchical conditional relation networks for video question answering. In CVPR."},{"key":"e_1_3_2_1_25_1","unstructured":"Kuang-Huei Lee Xi Chen Gang Hua Houdong Hu and Xiaodong He. 2018. Stacked cross attention for image-text matching. In ECCV."},{"key":"e_1_3_2_1_26_1","volume-title":"Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461","author":"Lewis Mike","year":"2019","unstructured":"Mike Lewis, Yinhan Liu, Naman Goyal, Marjan Ghazvininejad, Abdelrahman Mohamed, Omer Levy, Ves Stoyanov, and Luke Zettlemoyer. 2019. Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461 (2019)."},{"key":"e_1_3_2_1_27_1","unstructured":"Guangyao Li Yake Wei Yapeng Tian Chenliang Xu Ji-Rong Wen and Di Hu. 2022. Learning to answer questions in dynamic audio-visual scenarios. In CVPR."},{"key":"e_1_3_2_1_28_1","volume-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML.","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML."},{"key":"e_1_3_2_1_29_1","volume-title":"Align before fuse: Vision and language representation learning with momentum distillation. NIPS","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Ramprasaath Selvaraju, Akhilesh Gotmare, Shafiq Joty, Caiming Xiong, and Steven Chu Hong Hoi. 2021. Align before fuse: Vision and language representation learning with momentum distillation. NIPS (2021)."},{"key":"e_1_3_2_1_30_1","volume-title":"Hero: Hierarchical encoder for video language omni-representation pre-training. arXiv preprint arXiv:2005.00200","author":"Li Linjie","year":"2020","unstructured":"Linjie Li, Yen-Chun Chen, Yu Cheng, Zhe Gan, Licheng Yu, and Jingjing Liu. 2020. Hero: Hierarchical encoder for video language omni-representation pre-training. arXiv preprint arXiv:2005.00200 (2020)."},{"key":"e_1_3_2_1_31_1","unstructured":"Xiangpeng Li Jingkuan Song Lianli Gao Xianglong Liu Wenbing Huang Xiangnan He and Chuang Gan. 2019. Beyond rnns: Positional self-attention with co-attention for video question answering. In AAAI."},{"key":"e_1_3_2_1_32_1","volume-title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","author":"Li Xiujun","year":"2020","unstructured":"Xiujun Li, Xi Yin, Chunyuan Li, Pengchuan Zhang, Xiaowei Hu, Lei Zhang, Lijuan Wang, Houdong Hu, Li Dong, Furu Wei, et al. 2020. Oscar: Object-semantics aligned pre-training for vision-language tasks. In ECCV. Springer, 121--137."},{"key":"e_1_3_2_1_33_1","unstructured":"Yan-Bo Lin Yu-Jhe Li and Yu-Chiang Frank Wang. 2019. Dual-modality seq2seq network for audio-visual event localization. In ICASSP."},{"key":"e_1_3_2_1_34_1","volume-title":"Jie Lei, Mohit Bansal, and Gedas Bertasius.","author":"Lin Yi-Lin","year":"2023","unstructured":"Yi-Lin Lin, Yan-Bo an Sung, Jie Lei, Mohit Bansal, and Gedas Bertasius. 2023. Vision Transformers are Parameter-Efficient Audio-Visual Learners. In CVPR."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Ziyi Lin Shijie Geng Renrui Zhang Peng Gao Gerard de Melo Xiaogang Wang Jifeng Dai Yu Qiao and Hongsheng Li. 2022. Frozen clip models are efficient video learners. In ECCV.","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"e_1_3_2_1_36_1","volume-title":"Violin: A large-scale dataset for video-and-language inference. In CVPR.","author":"Liu Jingzhou","year":"2020","unstructured":"Jingzhou Liu, Wenhu Chen, Yu Cheng, Zhe Gan, Licheng Yu, Yiming Yang, and Jingjing Liu. 2020. Violin: A large-scale dataset for video-and-language inference. In CVPR."},{"key":"e_1_3_2_1_37_1","volume-title":"OPT: Omniperception pre-trainer for cross-modal understanding and generation. arXiv preprint arXiv:2107.00249","author":"Liu Jing","year":"2021","unstructured":"Jing Liu, Xinxin Zhu, Fei Liu, Longteng Guo, Zijia Zhao, Mingzhen Sun, Weining Wang, Hanqing Lu, Shiyu Zhou, Jiajun Zhang, et al. 2021. OPT: Omniperception pre-trainer for cross-modal understanding and generation. arXiv preprint arXiv:2107.00249 (2021)."},{"key":"e_1_3_2_1_38_1","volume-title":"Polyhistor: Parameter-Efficient Multi-Task Adaptation for Dense Vision Tasks. In NIPS.","author":"Liu Yen-Cheng","year":"2022","unstructured":"Yen-Cheng Liu, Chih-Yao Ma, Junjiao Tian, Zijian He, and Zsolt Kira. 2022. Polyhistor: Parameter-Efficient Multi-Task Adaptation for Dense Vision Tasks. In NIPS."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Ze Liu Yutong Lin Yue Cao Han Hu Yixuan Wei Zheng Zhang Stephen Lin and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_40_1","unstructured":"Ilya Loshchilov and Frank Hutter. 2019. Decoupled weight decay regularization. In ICLR."},{"key":"e_1_3_2_1_41_1","volume-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems (2019)."},{"key":"e_1_3_2_1_42_1","unstructured":"Jiasen Lu Caiming Xiong Devi Parikh and Richard Socher. 2017. Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In CVPR."},{"key":"e_1_3_2_1_43_1","volume-title":"Hierarchical question-image co-attention for visual question answering. NIPS","author":"Lu Jiasen","year":"2016","unstructured":"Jiasen Lu, Jianwei Yang, Dhruv Batra, and Devi Parikh. 2016. Hierarchical question-image co-attention for visual question answering. NIPS (2016)."},{"key":"e_1_3_2_1_44_1","volume-title":"Scalevlad: Improving multimodal sentiment analysis via multi-scale fusion of locally descriptors. arXiv preprint arXiv:2112.01368","author":"Luo Huaishao","year":"2021","unstructured":"Huaishao Luo, Lei Ji, Yanyong Huang, Bin Wang, Shenggong Ji, and Tianrui Li. 2021. Scalevlad: Improving multimodal sentiment analysis via multi-scale fusion of locally descriptors. arXiv preprint arXiv:2112.01368 (2021)."},{"key":"e_1_3_2_1_45_1","volume-title":"Active contrastive learning of audio-visual video representations. arXiv preprint arXiv:2009.09805","author":"Ma Shuang","year":"2020","unstructured":"Shuang Ma, Zhaoyang Zeng, Daniel McDuff, and Yale Song. 2020. Active contrastive learning of audio-visual video representations. arXiv preprint arXiv:2009.09805 (2020)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Tanvir Mahmud and Diana Marculescu. 2023. AVE-CLIP: AudioCLIP-based Multi-window Temporal Transformer for Audio Visual Event Localization. In WACV.","DOI":"10.1109\/WACV56688.2023.00513"},{"key":"e_1_3_2_1_47_1","volume-title":"Attention bottlenecks for multimodal fusion. NIPS","author":"Nagrani Arsha","year":"2021","unstructured":"Arsha Nagrani, Shan Yang, Anurag Arnab, Aren Jansen, Cordelia Schmid, and Chen Sun. 2021. Attention bottlenecks for multimodal fusion. NIPS (2021)."},{"key":"e_1_3_2_1_48_1","volume-title":"AdapterFusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247","author":"Pfeiffer Jonas","year":"2020","unstructured":"Jonas Pfeiffer, Aishwarya Kamath, Andreas R\u00fcckl\u00e9, Kyunghyun Cho, and Iryna Gurevych. 2020. AdapterFusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247 (2020)."},{"key":"e_1_3_2_1_49_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In ICML."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Janani Ramaswamy and Sukhendu Das. 2020. See the sound hear the pixels. In WACV.","DOI":"10.1109\/WACV45572.2020.9093616"},{"key":"e_1_3_2_1_51_1","volume-title":"Haoda Li, Peng Dai, and Juwei Lu.","author":"Rao Varshanth","year":"2022","unstructured":"Varshanth Rao, Md Ibrahim Khalil, Haoda Li, Peng Dai, and Juwei Lu. 2022. Dual Perspective Network for Audio-Visual Event Localization. In ECCV."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"crossref","unstructured":"Idan Schwartz Alexander G Schwing and Tamir Hazan. 2019. A simple baseline for audio-visual scene-aware dialog. In CVPR.","DOI":"10.1109\/CVPR.2019.01283"},{"key":"e_1_3_2_1_53_1","volume-title":"Hao Tan, Mohit Bansal, Anna Rohrbach, Kai-Wei Chang, Zhewei Yao, and Kurt Keutzer.","author":"Shen Sheng","year":"2021","unstructured":"Sheng Shen, Liunian Harold Li, Hao Tan, Mohit Bansal, Anna Rohrbach, Kai-Wei Chang, Zhewei Yao, and Kurt Keutzer. 2021. How much can clip benefit vision-and-language tasks? arXiv preprint arXiv:2107.06383 (2021)."},{"key":"e_1_3_2_1_54_1","volume-title":"Flava: A foundational language and vision alignment model. In CVPR.","author":"Singh Amanpreet","year":"2022","unstructured":"Amanpreet Singh, Ronghang Hu, Vedanuj Goswami, Guillaume Couairon, Wojciech Galuba, Marcus Rohrbach, and Douwe Kiela. 2022. Flava: A foundational language and vision alignment model. In CVPR."},{"key":"e_1_3_2_1_55_1","unstructured":"Zhongkai Sun Prathusha Sarma William Sethares and Yingyu Liang. 2020. Learn- ing relationships between text audio and video via deep canonical correlation for multimodal language analysis. In AAAI."},{"key":"e_1_3_2_1_56_1","volume-title":"Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In CVPR.","author":"Sung Yi-Lin","year":"2022","unstructured":"Yi-Lin Sung, Jaemin Cho, and Mohit Bansal. 2022. Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In CVPR."},{"key":"e_1_3_2_1_57_1","volume-title":"Lxmert: Learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490","author":"Tan Hao","year":"2019","unstructured":"Hao Tan and Mohit Bansal. 2019. Lxmert: Learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490 (2019)."},{"key":"e_1_3_2_1_58_1","volume-title":"TVLT: Textless Vision-Language Transformer. arXiv preprint arXiv:2209.14156","author":"Tang Zineng","year":"2022","unstructured":"Zineng Tang, Jaemin Cho, Yixin Nie, and Mohit Bansal. 2022. TVLT: Textless Vision-Language Transformer. arXiv preprint arXiv:2209.14156 (2022)."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746223"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"crossref","unstructured":"Yapeng Tian Di Hu and Chenliang Xu. 2021. Cyclic co-learning of sounding object visual grounding and sound separation. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00277"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"Yapeng Tian Jing Shi Bochen Li Zhiyao Duan and Chenliang Xu. 2018. Audio-visual event localization in unconstrained videos. In ECCV.","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"e_1_3_2_1_62_1","volume-title":"J Zico Kolter, Louis-Philippe Morency, and Ruslan Salakhutdinov.","author":"Hubert Tsai Yao-Hung","year":"2019","unstructured":"Yao-Hung Hubert Tsai, Shaojie Bai, Paul Pu Liang, J Zico Kolter, Louis-Philippe Morency, and Ruslan Salakhutdinov. 2019. Multimodal transformer for unaligned multimodal language sequences. In ACL."},{"key":"e_1_3_2_1_63_1","volume-title":"Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In ICML. PMLR.","author":"Wang Peng","year":"2022","unstructured":"Peng Wang, An Yang, Rui Men, Junyang Lin, Shuai Bai, Zhikang Li, Jianxin Ma, Chang Zhou, Jingren Zhou, and Hongxia Yang. 2022. Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In ICML. PMLR."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.bioorg.2022.105842"},{"key":"e_1_3_2_1_65_1","volume-title":"Saksham Singhal, Subhojit Som, et al.","author":"Wang Wenhui","year":"2022","unstructured":"Wenhui Wang, Hangbo Bao, Li Dong, Johan Bjorck, Zhiliang Peng, Qiang Liu, Kriti Aggarwal, Owais Khan Mohammed, Saksham Singhal, Subhojit Som, et al. 2022. Image as a foreign language: Beit pretraining for all vision and vision-language tasks. arXiv preprint arXiv:2208.10442 (2022)."},{"key":"e_1_3_2_1_66_1","volume-title":"Amir Zadeh, and Louis-Philippe Morency.","author":"Wang Yansen","year":"2019","unstructured":"Yansen Wang, Ying Shen, Zhun Liu, Paul Pu Liang, Amir Zadeh, and Louis-Philippe Morency. 2019. Words can shift: Dynamically adjusting word representations using nonverbal behaviors. In AAAI."},{"key":"e_1_3_2_1_67_1","unstructured":"Xuan Wu Qing-Guo Chen Yao Hu Dengbao Wang Xiaodong Chang Xiaobo Wang and Min-Ling Zhang. 2019. Multi-View Multi-Label Learning with View-Specific Information Extraction.. In IJCAI."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"crossref","unstructured":"Yan Xia and Zhou Zhao. 2022. Cross-modal background suppression for audio-visual event localization. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01936"},{"key":"e_1_3_2_1_69_1","unstructured":"Ziyi Yang Yuwei Fang Chenguang Zhu Reid Pryzant Dongdong Chen Yu Shi Yichong Xu Yao Qian Mei Gao Yi-Ling Chen et al. 2022. i-code: An integrative and composable multimodal learning framework. arXiv preprint arXiv:2205.01818 (2022)."},{"key":"e_1_3_2_1_70_1","volume-title":"Ernie-vil: Knowledge enhanced vision-language representations through scene graphs. In AAAI.","author":"Yu Fei","year":"2021","unstructured":"Fei Yu, Jiji Tang, Weichong Yin, Yu Sun, Hao Tian, Hua Wu, and Haifeng Wang. 2021. Ernie-vil: Knowledge enhanced vision-language representations through scene graphs. In AAAI."},{"key":"e_1_3_2_1_71_1","unstructured":"Wenmeng Yu Hua Xu Ziqi Yuan and Jiele Wu. 2021. Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. In AAAI."},{"key":"e_1_3_2_1_72_1","unstructured":"Zhou Yu Jun Yu Yuhao Cui Dacheng Tao and Qi Tian. 2019. Deep modular co-attention networks for visual question answering. In CVPR."},{"key":"e_1_3_2_1_73_1","volume-title":"Pano-avqa: Grounded audio-visual question answering on 360deg videos. In ICCV.","author":"Yun Heeseung","year":"2021","unstructured":"Heeseung Yun, Youngjae Yu, Wonsuk Yang, Kangil Lee, and Gunhee Kim. 2021. Pano-avqa: Grounded audio-visual question answering on 360deg videos. In ICCV."},{"key":"e_1_3_2_1_74_1","volume-title":"Soujanya Poria, Erik Cambria, and Louis-Philippe Morency.","author":"Bagher Zadeh AmirAli","year":"2018","unstructured":"AmirAli Bagher Zadeh, Paul Pu Liang, Soujanya Poria, Erik Cambria, and Louis-Philippe Morency. 2018. Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph. In ACL."},{"key":"e_1_3_2_1_75_1","volume-title":"Soujanya Poria, Erik Cambria, and Louis-Philippe Morency.","author":"Bagher Zadeh AmirAli","year":"2018","unstructured":"AmirAli Bagher Zadeh, Paul Pu Liang, Soujanya Poria, Erik Cambria, and Louis-Philippe Morency. 2018. Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph. In ACL."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"crossref","unstructured":"Rowan Zellers Jiasen Lu Ximing Lu Youngjae Yu Yanpeng Zhao Mohammadreza Salehi Aditya Kusupati Jack Hessel Ali Farhadi and Yejin Choi. 2022. Merlot reserve: Neural script knowledge through vision and language and sound. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01589"},{"key":"e_1_3_2_1_77_1","volume-title":"Multi-grained vision language pre-training: Aligning texts with visual concepts. arXiv preprint arXiv:2111.08276","author":"Zeng Yan","year":"2021","unstructured":"Yan Zeng, Xinsong Zhang, and Hang Li. 2021. Multi-grained vision language pre-training: Aligning texts with visual concepts. arXiv preprint arXiv:2111.08276 (2021)."},{"key":"e_1_3_2_1_78_1","volume-title":"X2-VLM: All-In-One Pre-trained Model For Vision-Language Tasks. arXiv preprint arXiv:2211.12402","author":"Zeng Yan","year":"2022","unstructured":"Yan Zeng, Xinsong Zhang, Hang Li, Jiawei Wang, Jipeng Zhang, and Wangchun-shu Zhou. 2022. X2-VLM: All-In-One Pre-trained Model For Vision-Language Tasks. arXiv preprint arXiv:2211.12402 (2022)."},{"key":"e_1_3_2_1_79_1","unstructured":"Zhaoyang Zeng Daniel McDuff Yale Song et al. 2021. Contrastive learning of global and local video representations. NIPS (2021)."},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"crossref","unstructured":"Dong Zhang Xincheng Ju Wei Zhang Junhui Li Shoushan Li Qiaoming Zhu and Guodong Zhou. 2021. Multi-modal multi-label emotion recognition with heterogeneous hierarchical message passing. In AAAI.","DOI":"10.1609\/aaai.v35i16.17686"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_29"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611939","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3611939","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:04:32Z","timestamp":1755821072000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611939"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":81,"alternative-id":["10.1145\/3581783.3611939","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3611939","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}