{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T09:53:15Z","timestamp":1761126795220,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National NaturalScience Foundation of China","award":["62006151, 62076161"],"award-info":[{"award-number":["62006151, 62076161"]}]},{"name":"Shanghai Sailing Program"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475366","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T21:45:34Z","timestamp":1634593534000},"page":"2103-2111","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Deconfounded and Explainable Interactive Vision-Language Retrieval of Complex Scenes"],"prefix":"10.1145","author":[{"given":"Junda","family":"Wu","sequence":"first","affiliation":[{"name":"New York University, New York City, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tong","family":"Yu","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuai","family":"Li","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10044--10054","author":"Abbasnejad Ehsan","key":"e_1_3_2_1_1_1"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3109859.3109913"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00636"},{"volume-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. 1140--1149","year":"2020","author":"Anwaar Muhammad Umer","key":"e_1_3_2_1_4_1"},{"volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","year":"2016","author":"Ba Jimmy Lei","key":"e_1_3_2_1_5_1"},{"volume-title":"International Workshop on Adaptive Multimedia Retrieval. Springer, 32--44","year":"2007","author":"Bartolini Ilaria","key":"e_1_3_2_1_6_1"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1385569.1385664"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.5555\/645927.672026"},{"volume-title":"Proceedings of the International Conference on Signal Processing and Multimedia Applications. IEEE, 1--10","year":"2011","author":"Bartolini Ilaria","key":"e_1_3_2_1_9_1"},{"volume-title":"Generate natural language explanations for recommendation. arXiv preprint arXiv:2101.03392","year":"2021","author":"Chen Hanxiong","key":"e_1_3_2_1_10_1"},{"volume-title":"Visually explainable recommendation. arXiv preprint arXiv:1801.10288","year":"2018","author":"Chen Xu","key":"e_1_3_2_1_11_1"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00307"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/3367243.3367336"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Devlin Jacob","year":"2019"},{"volume-title":"CVPR Workshops. 95--98","year":"2019","author":"Dong Bo","key":"e_1_3_2_1_15_1"},{"volume-title":"International Conference on Machine Learning. 2454--2463","year":"2019","author":"Guan Chaoyu","key":"e_1_3_2_1_16_1"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.5555\/3326943.3327006"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/3367471.3367694"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-1082"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3411992"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Zhi Li Bo Wu Qi Liu Likang Wu Hongke Zhao and Tao Mei. 2020 a. Learning the Compositional Visual Coherence for Complementary Recommendations. In IJCAI .  Zhi Li Bo Wu Qi Liu Likang Wu Hongke Zhao and Tao Mei. 2020 a. Learning the Compositional Visual Coherence for Complementary Recommendations. In IJCAI .","DOI":"10.24963\/ijcai.2020\/489"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240646"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2019.102099"},{"volume-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. arXiv preprint arXiv:1908.02265","year":"2019","author":"Lu Jiasen","key":"e_1_3_2_1_26_1"},{"volume-title":"Hierarchical Similarity Learning for Language-based Product Image Retrieval. arXiv preprint arXiv:2102.09375","year":"2021","author":"Ma Zhe","key":"e_1_3_2_1_27_1"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01251"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01087"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018661.3018686"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.5555\/645923.673655"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/34.895972"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3454525"},{"volume-title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models. arXiv preprint arXiv:1908.08962v2","year":"2019","author":"Turc Iulia","key":"e_1_3_2_1_35_1"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3358028"},{"volume-title":"Gaurav Singh Tomar, and Manaal Faruqui","year":"2019","author":"Vashishth Shikhar","key":"e_1_3_2_1_37_1"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01077"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01115"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00972"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6485"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330991"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350935"},{"volume-title":"2020 b. Reward Constrained Interactive Recommendation with Natural Language Feedback. arXiv preprint arXiv:2005.01618","year":"2020","author":"Zhang Ruiyi","key":"e_1_3_2_1_44_1"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413518"},{"volume-title":"Explainable recommendation: A survey and new perspectives. arXiv preprint arXiv:1804.11192","year":"2018","author":"Zhang Yongfeng","key":"e_1_3_2_1_46_1"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600428.2609579"}],"event":{"name":"MM '21: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Virtual Event China","acronym":"MM '21"},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475366","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475366","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:49:19Z","timestamp":1750193359000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475366"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":47,"alternative-id":["10.1145\/3474085.3475366","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475366","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}