{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T09:04:26Z","timestamp":1765357466076,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":71,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62250008, 62222209, 62102222"],"award-info":[{"award-number":["62250008, 62222209, 62102222"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Program of China","award":["2020AAA0106300"],"award-info":[{"award-number":["2020AAA0106300"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612401","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"4450-4459","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["Mixup-Augmented Temporally Debiased Video Grounding with Content-Location Disentanglement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0351-2939","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"first","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4389-2980","authenticated-orcid":false,"given":"Zihao","family":"Wu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0943-2286","authenticated-orcid":false,"given":"Hong","family":"Chen","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5382-6699","authenticated-orcid":false,"given":"Xiaohan","family":"Lan","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2236-9290","authenticated-orcid":false,"given":"Wenwu","family":"Zhu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512527.3531403"},{"key":"e_1_3_2_1_2_1","volume-title":"Representation learning: A review and new perspectives","author":"Bengio Yoshua","year":"2013","unstructured":"Yoshua Bengio, Aaron Courville, and Pascal Vincent. 2013. Representation learning: A review and new perspectives. IEEE transactions on pattern analysis and machine intelligence, Vol. 35, 8 (2013), 1798--1828."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11867"},{"key":"e_1_3_2_1_4_1","volume-title":"Understanding disentangling in \u03b2-VAE. arXiv preprint arXiv:1804.03599","author":"Burgess Christopher P","year":"2018","unstructured":"Christopher P Burgess, Irina Higgins, Arka Pal, Loic Matthey, Nick Watters, Guillaume Desjardins, and Alexander Lerchner. 2018. Understanding disentangling in \u03b2-VAE. arXiv preprint arXiv:1804.03599 (2018)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_6_1","first-page":"26924","article-title":"Curriculum disentangled recommendation with noisy multi-feedback","volume":"34","author":"Chen Hong","year":"2021","unstructured":"Hong Chen, Yudong Chen, Xin Wang, Ruobing Xie, Rui Wang, Feng Xia, and Wenwu Zhu. 2021. Curriculum disentangled recommendation with noisy multi-feedback. Advances in Neural Information Processing Systems, Vol. 34 (2021), 26924--26936.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.194"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6627"},{"key":"e_1_3_2_1_9_1","unstructured":"Tian Qi Chen Xuechen Li Roger B Grosse and David K Duvenaud. 2018. Isolating sources of disentanglement in variational autoencoders. In Advances in Neural Information Processing Systems. 2610--2620."},{"key":"e_1_3_2_1_10_1","volume-title":"Infogan: Interpretable representation learning by information maximizing generative adversarial nets. In Advances in neural information processing systems. 2172--2180.","author":"Chen Xi","year":"2016","unstructured":"Xi Chen, Yan Duan, Rein Houthooft, John Schulman, Ilya Sutskever, and Pieter Abbeel. 2016. Infogan: Interpretable representation learning by information maximizing generative adversarial nets. In Advances in neural information processing systems. 2172--2180."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.529"},{"key":"e_1_3_2_1_12_1","volume-title":"Marta Garnelo, Matthew CH Lee, Hugh Salimbeni, Kai Arulkumaran, and Murray Shanahan.","author":"Dilokthanakul Nat","year":"2016","unstructured":"Nat Dilokthanakul, Pedro AM Mediano, Marta Garnelo, Matthew CH Lee, Hugh Salimbeni, Kai Arulkumaran, and Murray Shanahan. 2016. Deep unsupervised clustering with gaussian mixture variational autoencoders. arXiv preprint arXiv:1611.02648 (2016)."},{"key":"e_1_3_2_1_13_1","unstructured":"Emilien Dupont. 2018. Learning disentangled joint continuous and discrete representations. In Advances in Neural Information Processing Systems. 710--720."},{"key":"e_1_3_2_1_14_1","volume-title":"Frederic Besse, Fabio Viola, Ari S Morcos, Marta Garnelo, Avraham Ruderman, Andrei A Rusu, Ivo Danihelka, Karol Gregor, et al.","author":"Ali Eslami SM","year":"2018","unstructured":"SM Ali Eslami, Danilo Jimenez Rezende, Frederic Besse, Fabio Viola, Ari S Morcos, Marta Garnelo, Avraham Ruderman, Andrei A Rusu, Ivo Danihelka, Karol Gregor, et al. 2018. Neural scene representation and rendering. Science, Vol. 360, 6394 (2018), 1204--1210."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Steven Y Feng Varun Gangal Jason Wei Sarath Chandar Soroush Vosoughi Teruko Mitamura and Eduard Hovy. 2021. A survey of data augmentation approaches for nlp. (2021) 968--988.","DOI":"10.18653\/v1\/2021.findings-acl.84"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"e_1_3_2_1_17_1","volume-title":"International Conference on Learning Representations.","author":"Ghandeharioun Asma","year":"2021","unstructured":"Asma Ghandeharioun, Been Kim, Chun-Liang Li, Brendan Jou, Brian Eoff, and Rosalind Picard. 2021. DISSECT: Disentangled Simultaneous Explanations via Concept Traversals. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5822"},{"key":"e_1_3_2_1_19_1","volume-title":"Augmenting data with mixup for sentence classification: An empirical study. arXiv preprint arXiv:1905.08941","author":"Guo Hongyu","year":"2019","unstructured":"Hongyu Guo, Yongyi Mao, and Richong Zhang. 2019. Augmenting data with mixup for sentence classification: An empirical study. arXiv preprint arXiv:1905.08941 (2019)."},{"key":"e_1_3_2_1_20_1","volume-title":"31st British Machine Vision Conference 2020, BMVC 2020","author":"Hahn Meera","year":"2019","unstructured":"Meera Hahn, Asim Kadav, James M Rehg, and Hans Peter Graf. 2019. Tripping through time: Efficient localization of activities in videos. In 31st British Machine Vision Conference 2020, BMVC 2020, Virtual Event, UK, September 7-10, 2020."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1036"},{"key":"e_1_3_2_1_22_1","volume-title":"International Conference on Learning Representations","volume":"3","author":"Higgins Irina","year":"2017","unstructured":"Irina Higgins, Loic Matthey, Arka Pal, Christopher Burgess, Xavier Glorot, Matthew Botvinick, Shakir Mohamed, and Alexander Lerchner. 2017. \u03b2-vae: Learning basic visual concepts with a constrained variational framework. In International Conference on Learning Representations, Vol. 3."},{"key":"e_1_3_2_1_23_1","unstructured":"Jun-Ting Hsieh Bingbin Liu De-An Huang Li F Fei-Fei and Juan Carlos Niebles. 2018. Learning to decompose and disentangle representations for video prediction. In Advances in Neural Information Processing Systems. 517--526."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.493"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00711"},{"volume-title":"Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing. 4683--4693","author":"Jain Sarthak","key":"e_1_3_2_1_26_1","unstructured":"Sarthak Jain, Edward Banner, Jan-Willem van de Meent, Iain Marshall, and Byron C Wallace. 2018. Learning Disentangled Representations of Texts with Application to Biomedical Abstracts. In Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing. 4683--4693."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3323873.3325019"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.5555\/3172077.3172161"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3147"},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Machine Learning. 2654--2663","author":"Kim Hyunjik","year":"2018","unstructured":"Hyunjik Kim and Andriy Mnih. 2018. Disentangling by Factorising. In International Conference on Machine Learning. 2654--2663."},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Max Welling. 2014. Auto-encoding variational bayes. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_32_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Komodakis Nikos","year":"2018","unstructured":"Nikos Komodakis and Spyros Gidaris. 2018. Unsupervised representation learning by predicting image rotations. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_33_1","volume-title":"Yee Whye Teh, and Ingmar Posner","author":"Kosiorek Adam","year":"2018","unstructured":"Adam Kosiorek, Hyunjik Kim, Yee Whye Teh, and Ingmar Posner. 2018. Sequential attend, infer, repeat: Generative modelling of moving objects. In Advances in Neural Information Processing Systems. 8606--8616."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25204"},{"key":"e_1_3_2_1_36_1","volume-title":"A closer look at debiased temporal sentence grounding in videos: Dataset, metric, and approach. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM)","author":"Lan Xiaohan","year":"2022","unstructured":"Xiaohan Lan, Yitian Yuan, Xin Wang, Long Chen, Zhi Wang, Lin Ma, and Wenwu Zhu. 2022. A closer look at debiased temporal sentence grounding in videos: Dataset, metric, and approach. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) (2022)."},{"key":"e_1_3_2_1_37_1","volume-title":"A survey on temporal sentence grounding in videos. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM)","author":"Lan Xiaohan","year":"2021","unstructured":"Xiaohan Lan, Yitian Yuan, Xin Wang, Zhi Wang, and Wenwu Zhu. 2021. A survey on temporal sentence grounding in videos. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) (2021)."},{"key":"e_1_3_2_1_38_1","first-page":"21872","article-title":"Disentangled contrastive learning on graphs","volume":"34","author":"Li Haoyang","year":"2021","unstructured":"Haoyang Li, Xin Wang, Ziwei Zhang, Zehuan Yuan, Hang Li, and Wenwu Zhu. 2021. Disentangled contrastive learning on graphs. Advances in Neural Information Processing Systems, Vol. 34 (2021), 21872--21884.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00304"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548035"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210003"},{"key":"e_1_3_2_1_42_1","volume-title":"International conference on machine learning. PMLR, 4212--4221","author":"Ma Jianxin","year":"2019","unstructured":"Jianxin Ma, Peng Cui, Kun Kuang, Xin Wang, and Wenwu Zhu. 2019a. Disentangled graph convolutional networks. In International conference on machine learning. PMLR, 4212--4221."},{"key":"e_1_3_2_1_43_1","volume-title":"Learning disentangled representations for recommendation. Advances in neural information processing systems","author":"Ma Jianxin","year":"2019","unstructured":"Jianxin Ma, Chang Zhou, Peng Cui, Hongxia Yang, and Wenwu Zhu. 2019b. Learning disentangled representations for recommendation. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00018"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-57321-8_21"},{"key":"e_1_3_2_1_46_1","volume-title":"Uncovering Hidden Challenges in Query-Based Video Moment Retrieval. In The British Machine Vision Conference (BMVC).","author":"Mayu Otani Esa Rahtu","year":"2020","unstructured":"Esa Rahtu Mayu Otani, Yuta Nakahima and Janne Heikkil\u00e4. 2020. Uncovering Hidden Challenges in Query-Based Video Moment Retrieval. In The British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00279"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00207"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.305"},{"key":"e_1_3_2_1_51_1","volume-title":"Proc. of the 37th Allerton Conference on Communication and Computation","author":"N","year":"1999","unstructured":"N TISHBY. 1999. The Information Bottleneck Method. In Proc. of the 37th Allerton Conference on Communication and Computation, 1999."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ITW.2015.7133169"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3153112"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME51207.2021.9428193"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401137"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6924"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.327"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16406"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462823"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3475723.3484247"},{"key":"e_1_3_2_1_62_1","volume-title":"Advances in Neural Information Processing Systems","volume":"32","author":"Yuan Yitian","year":"2019","unstructured":"Yitian Yuan, Lin Ma, Jingwen Wang, Wei Liu, and Wenwu Zhu. 2019a. Semantic conditioned dynamic modulation for temporal sentence grounding in videos. Advances in Neural Information Processing Systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"e_1_3_2_1_65_1","volume-title":"International Conference on Learning Representations.","author":"Zhang Hongyi","year":"2018","unstructured":"Hongyi Zhang, Moustapha Cisse, Yann N Dauphin, and David Lopez-Paz. 2018. mixup: Beyond Empirical Risk Minimization. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383313.3412239"},{"key":"e_1_3_2_1_69_1","first-page":"18123","article-title":"Counterfactual contrastive learning for weakly-supervised vision-language grounding","volume":"33","author":"Zhang Zhu","year":"2020","unstructured":"Zhu Zhang, Zhou Zhao, Zhijie Lin, Xiuqiang He, et al. 2020c. Counterfactual contrastive learning for weakly-supervised vision-language grounding. Advances in Neural Information Processing Systems, Vol. 33 (2020), 18123--18134.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00834"},{"key":"e_1_3_2_1_71_1","unstructured":"Jun-Yan Zhu Zhoutong Zhang Chengkai Zhang Jiajun Wu Antonio Torralba Josh Tenenbaum and Bill Freeman. 2018. Visual object networks: Image generation with disentangled 3D representations. In Advances in Neural Information Processing Systems. 118--129."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612401","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612401","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:02:12Z","timestamp":1755820932000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612401"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":71,"alternative-id":["10.1145\/3581783.3612401","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612401","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}