{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:21:37Z","timestamp":1765308097728,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China under Grant","award":["62325206, 62036012, U23A20387, 62322212"],"award-info":[{"award-number":["62325206, 62036012, U23A20387, 62322212"]}]},{"name":"the Key Research and Development Program of Jiangsu Province under Grant","award":["BE2023016-4"],"award-info":[{"award-number":["BE2023016-4"]}]},{"name":"Pengcheng Laboratory Research Project under Grant","award":["PCL2023A08"],"award-info":[{"award-number":["PCL2023A08"]}]},{"name":"Postdoctoral Fellowship Program of CPSF under Grant Number","award":["GZC20251036"],"award-info":[{"award-number":["GZC20251036"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755085","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"3438-3447","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["DMC\n                    <sup>3<\/sup>\n                    : Dual-Modal Counterfactual Contrastive Construction for Egocentric Video Question Answering"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-8821-6358","authenticated-orcid":false,"given":"Jiayi","family":"Zou","sequence":"first","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7970-7698","authenticated-orcid":false,"given":"Chaofan","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5956-831X","authenticated-orcid":false,"given":"Bing-Kun","family":"Bao","sequence":"additional","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China and Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8343-9665","authenticated-orcid":false,"given":"Changsheng","family":"Xu","sequence":"additional","affiliation":[{"name":"MAIS, Institute of Automation, Chinese Academy of Sciences, Beijing, China and Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"65","volume-title":"Proceedings of ACL-WMT","author":"Banerjee Satanjeev","year":"2004","unstructured":"Satanjeev Banerjee and Alon Lavie. 2004. Meteor: an automatic metric for mt evaluation with high levels of correlation with human judgments. Proceedings of ACL-WMT (2004), 65-72."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00162"},{"key":"e_1_3_2_1_3_1","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is space-time attention all you need for video understanding?. In ICML, Vol. 2. 4.","journal-title":"ICML"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681214"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01081"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3290012"},{"key":"e_1_3_2_1_7_1","unstructured":"Qirui Chen Shangzhe Di and Weidi Xie. 2024a. Grounded Multi-Hop VideoQA in Long-Form Egocentric Videos. arXiv:2408.14469 [cs.CV] https:\/\/arxiv.org\/abs\/2408.14469"},{"key":"e_1_3_2_1_8_1","volume-title":"GPT4Ego: unleashing the potential of pre-trained models for zero-shot egocentric action recognition","author":"Dai Guangzhao","year":"2024","unstructured":"Guangzhao Dai, Xiangbo Shu, Wenhao Wu, Rui Yan, and Jiachao Zhang. 2024. GPT4Ego: unleashing the potential of pre-trained models for zero-shot egocentric action recognition. IEEE Transactions on Multimedia (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01531-2"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01229"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548365"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00536"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_1_14_1","unstructured":"Longteng Guo Jing Liu Xinxin Zhu Xingjian He Jie Jiang and Hanqing Lu. 2020. Non-Autoregressive Image Captioning with Counterfactuals-Critical Multi-Agent Learning. arXiv:2005.04690 [cs.CL] https:\/\/arxiv.org\/abs\/2005.04690"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu et al. 2022. A survey on vision transformer. IEEE transactions on pattern analysis and machine intelligence Vol. 45 1 (2022) 87-110.","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02084"},{"key":"e_1_3_2_1_17_1","volume-title":"European Conference on Computer Vision. Springer, 767-786","author":"Jia Baoxiong","year":"2020","unstructured":"Baoxiong Jia, Yixin Chen, Siyuan Huang, Yixin Zhu, and Song-chun Zhu. 2020. LEMMA: A Multi-view Dataset for LE arning M ulti-agent M ulti-task A ctivities. In European Conference on Computer Vision. Springer, 767-786."},{"key":"e_1_3_2_1_18_1","first-page":"3343","article-title":"Egotaskqa: Understanding human tasks in egocentric videos","volume":"35","author":"Jia Baoxiong","year":"2022","unstructured":"Baoxiong Jia, Ting Lei, Song-Chun Zhu, and Siyuan Huang. 2022. Egotaskqa: Understanding human tasks in egocentric videos. Advances in Neural Information Processing Systems, Vol. 35 (2022), 3343-3360.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6767"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i7.25983"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICACITE51222.2021.9404712"},{"key":"e_1_3_2_1_22_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01514-3"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612395"},{"key":"e_1_3_2_1_26_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_1_27_1","first-page":"7575","article-title":"Egocentric video-language pretraining","volume":"35","author":"Lin Kevin Qinghong","year":"2022","unstructured":"Kevin Qinghong Lin, Jinpeng Wang, Mattia Soldan, Michael Wray, Rui Yan, Eric Z Xu, Difei Gao, Rong-Cheng Tu, Wenzhe Zhao, Weijie Kong, et al., 2022. Egocentric video-language pretraining. Advances in Neural Information Processing Systems, Vol. 35 (2022), 7575-7586.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3251108"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612239"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3284038"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612388"},{"key":"e_1_3_2_1_32_1","first-page":"655","article-title":"Where are we in the search for an artificial visual cortex for embodied intelligence","volume":"36","author":"Majumdar Arjun","year":"2023","unstructured":"Arjun Majumdar, Karmesh Yadav, Sergio Arnaud, Jason Ma, Claire Chen, Sneha Silwal, Aryan Jain, Vincent-Pierre Berges, Tingfan Wu, Jay Vakil, et al., 2023. Where are we in the search for an artificial visual cortex for embodied intelligence? Advances in Neural Information Processing Systems, Vol. 36 (2023), 655-677.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_33_1","volume-title":"International conference on Machine learning. PMLR, 23803-23828","author":"Mao Anqi","year":"2023","unstructured":"Anqi Mao, Mehryar Mohri, and Yutao Zhong. 2023. Cross-entropy loss functions: Theoretical analysis and applications. In International conference on Machine learning. PMLR, 23803-23828."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103862"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01257"},{"key":"e_1_3_2_1_36_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Mu Yao","year":"2024","unstructured":"Yao Mu, Qinglong Zhang, Mengkang Hu, Wenhai Wang, Mingyu Ding, Jun Jin, Bin Wang, Jifeng Dai, Yu Qiao, and Ping Luo. 2024. Embodiedgpt: Vision-language pre-training via embodied chain of thought. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_37_1","first-page":"60130","article-title":"EgoEnv: Human-centric environment representations from egocentric video","volume":"36","author":"Nagarajan Tushar","year":"2023","unstructured":"Tushar Nagarajan, Santhosh Kumar Ramakrishnan, Ruta Desai, James Hillis, and Kristen Grauman. 2023. EgoEnv: Human-centric environment representations from egocentric video. Advances in Neural Information Processing Systems, Vol. 36 (2023), 60130-60143.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_38_1","volume-title":"Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_1_39_1","volume-title":"Egovideo: Exploring egocentric foundation model and downstream adaptation. arXiv preprint arXiv:2406.18070","author":"Pei Baoqi","year":"2024","unstructured":"Baoqi Pei, Guo Chen, Jilan Xu, Yuping He, Yicheng Liu, Kanghua Pan, Yifei Huang, Yali Wang, Tong Lu, Limin Wang, et al., 2024. Egovideo: Exploring egocentric foundation model and downstream adaptation. arXiv preprint arXiv:2406.18070 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Kashu Yamazaki Minh Tran, and Ngan Le","author":"Thinh Phan Khoa Vo","year":"2024","unstructured":"Khoa Vo Thinh Phan, Kashu Yamazaki Minh Tran, and Ngan Le. 2024. HENASY: Learning to Assemble Scene-Entities for Interpretable Egocentric Video-Language Model. Advances in Neural Information Processing Systems (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"Dima Damen, and Tatiana Tommasi.","author":"Plizzari Chiara","year":"2024","unstructured":"Chiara Plizzari, Gabriele Goletto, Antonino Furnari, Siddhant Bansal, Francesco Ragusa, Giovanni Maria Farinella, Dima Damen, and Tatiana Tommasi. 2024. An outlook into the future of egocentric vision. International Journal of Computer Vision (2024), 1-57."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00487"},{"key":"e_1_3_2_1_43_1","volume-title":"Embodied intelligence toward future smart manufacturing in the era of AI foundation model","author":"Ren Lei","year":"2024","unstructured":"Lei Ren, Jiabao Dong, Shuai Liu, Lin Zhang, and Lihui Wang. 2024. Embodied intelligence toward future smart manufacturing in the era of AI foundation model. IEEE\/ASME Transactions on Mechatronics (2024)."},{"key":"e_1_3_2_1_44_1","first-page":"137372","article-title":"ActionAtlas: A VideoQA Benchmark for Domain-specialized Action Recognition","volume":"37","author":"Salehi Mohammadreza Reza","year":"2024","unstructured":"Mohammadreza Reza Salehi, Jae Sung Park, Aditya Kusupati, Ranjay Krishna, Yejin Choi, Hanna Hajishirzi, and Ali Farhadi. 2024. ActionAtlas: A VideoQA Benchmark for Domain-specialized Action Recognition. Advances in Neural Information Processing Systems, Vol. 37 (2024), 137372-137402.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_45_1","first-page":"338","article-title":"A replicable comparison study of NER software: StanfordNLP, NLTK, OpenNLP, SpaCy, Gate. In 2019 sixth international conference on social networks analysis, management and security (SNAMS)","author":"Schmitt Xavier","year":"2019","unstructured":"Xavier Schmitt, Sylvain Kubler, J\u00e9r\u00e9my Robert, Mike Papadakis, and Yves LeTraon. 2019. A replicable comparison study of NER software: StanfordNLP, NLTK, OpenNLP, SpaCy, Gate. In 2019 sixth international conference on social networks analysis, management and security (SNAMS). IEEE, 338-343.","journal-title":"IEEE"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Dandan Shan Jiaqi Geng Michelle Shu and David Fouhey. 2020. Understanding Human Hands in Contact at Internet Scale.","DOI":"10.1109\/CVPR42600.2020.00989"},{"key":"e_1_3_2_1_47_1","volume-title":"What makes for good views for contrastive learning? Advances in neural information processing systems","author":"Tian Yonglong","year":"2020","unstructured":"Yonglong Tian, Chen Sun, Ben Poole, Dilip Krishnan, Cordelia Schmid, and Phillip Isola. 2020. What makes for good views for contrastive learning? Advances in neural information processing systems, Vol. 33 (2020), 6827-6839."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00484"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00638"},{"key":"e_1_3_2_1_50_1","volume-title":"European Conference on Computer Vision. Springer, 58-76","author":"Wang Xiaohan","year":"2024","unstructured":"Xiaohan Wang, Yuhui Zhang, Orr Zohar, and Serena Yeung-Levy. 2024. Videoagent: Long-form video understanding with large language model as agent. In European Conference on Computer Vision. Springer, 58-76."},{"key":"e_1_3_2_1_51_1","volume-title":"EDA: Easy data augmentation techniques for boosting performance on text classification tasks. arXiv preprint arXiv:1901.11196","author":"Wei Jason","year":"2019","unstructured":"Jason Wei and Kai Zou. 2019. EDA: Easy data augmentation techniques for boosting performance on text classification tasks. arXiv preprint arXiv:1901.11196 (2019)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.432"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20184"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3330070"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2024.3521725"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01278"},{"key":"e_1_3_2_1_57_1","volume-title":"Multi-Factor Adaptive Vision Selection for Egocentric Video Question Answering. In Forty-first International Conference on Machine Learning.","author":"Zhang Haoyu","year":"2024","unstructured":"Haoyu Zhang, Meng Liu, Zixin Liu, Xuemeng Song, Yaowei Wang, and Liqiang Nie. 2024a. Multi-Factor Adaptive Vision Selection for Egocentric Video Question Answering. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02064"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475328"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"e_1_3_2_1_61_1","volume-title":"EgoTextVQA: Towards Egocentric Scene-Text Aware Video Question Answering. arXiv preprint arXiv:2502.07411","author":"Zhou Sheng","year":"2025","unstructured":"Sheng Zhou, Junbin Xiao, Qingyun Li, Yicong Li, Xun Yang, Dan Guo, Meng Wang, Tat-Seng Chua, and Angela Yao. 2025. EgoTextVQA: Towards Egocentric Scene-Text Aware Video Question Answering. arXiv preprint arXiv:2502.07411 (2025)."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02560"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888014"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755085","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:19:45Z","timestamp":1765307985000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755085"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":63,"alternative-id":["10.1145\/3746027.3755085","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755085","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}