{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:21:02Z","timestamp":1765308062650,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755492","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:47:42Z","timestamp":1761371262000},"page":"4534-4541","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Compositional Zero-Shot Learning with Contextualized Cues and Adaptive Contrastive Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4442-3825","authenticated-orcid":false,"given":"Yun","family":"Li","sequence":"first","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4149-839X","authenticated-orcid":false,"given":"Lina","family":"Yao","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2692-2110","authenticated-orcid":false,"given":"Zhe","family":"Liu","sequence":"additional","affiliation":[{"name":"Oracle, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Prompting Language-Informed Distribution for Compositional Zero-Shot Learning. arXiv preprint arXiv:2305.14428","author":"Bao Wentao","year":"2023","unstructured":"Wentao Bao, Lichang Chen, Heng Huang, and Yu Kong. 2023. Prompting Language-Informed Distribution for Compositional Zero-Shot Learning. arXiv preprint arXiv:2305.14428 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Doveh Sivan","year":"2024","unstructured":"Sivan Doveh, Assaf Arbelle, Sivan Harary, Roei Herzig, Donghyun Kim, Paola Cascante-Bonilla, Amit Alfassy, Rameswar Panda, Raja Giryes, Rogerio Feris, et al., 2024. Dense and aligned captions (dac) promote compositional reasoning in vl models. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_3_1","volume-title":"SugarCrepe: Fixing Hackable Benchmarks for Vision-Language Compositionality. In Thirty-Seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Hsieh Cheng-Yu","year":"2023","unstructured":"Cheng-Yu Hsieh, Jieyu Zhang, Zixian Ma, Aniruddha Kembhavi, and Ranjay Krishna. 2023. SugarCrepe: Fixing Hackable Benchmarks for Vision-Language Compositionality. In Thirty-Seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02266"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298744"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.28026"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00912"},{"key":"e_1_3_2_1_8_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_9_1","unstructured":"Brenden M Lake. 2014. Towards more human-like concept learning in machines: Compositionality causality and learning-to-learn. Ph.D. Dissertation. Massachusetts Institute of Technology."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00911"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01612"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00171"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01133"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3323012"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02256"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00518"},{"key":"e_1_3_2_1_17_1","volume-title":"Yongqin Xian, and Zeynep Akata.","author":"Mancini Massimiliano","year":"2022","unstructured":"Massimiliano Mancini, Muhammad Ferjad Naeem, Yongqin Xian, and Zeynep Akata. 2022. Learning graph embeddings for open world compositional zero-shot learning. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01428"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00101"},{"volume-title":"International Conference on Learning Representations.","author":"Nayak Nihal V.","key":"e_1_3_2_1_20_1","unstructured":"Nihal V. Nayak, Peilin Yu, and Stephen H. Bach. 2023. Learning to Compose Soft Prompts for Compositional Zero-Shot Learning. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00369"},{"key":"e_1_3_2_1_22_1","volume-title":"International conference on machine learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748-8763."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01329"},{"key":"e_1_3_2_1_24_1","volume-title":"Coarse-to-Fine Contrastive Learning in Image-Text-Graph Space for Improved Vision-Language Compositionality. arXiv preprint arXiv:2305.13812","author":"Singh Harman","year":"2023","unstructured":"Harman Singh, Pengchuan Zhang, Qifan Wang, Mengjiao Wang, Wenhan Xiong, Jingfei Du, and Yu Chen. 2023. Coarse-to-Fine Contrastive Learning in Image-Text-Graph Space for Improved Vision-Language Compositionality. arXiv preprint arXiv:2305.13812 (2023)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00517"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Henan Wang Muli Yang Kun Wei and Cheng Deng. 2023b. Hierarchical Prompt Learning for Compositional Zero-Shot Recognition. In IJCAI.","DOI":"10.24963\/ijcai.2023\/163"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01077"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00384"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00567"},{"key":"e_1_3_2_1_30_1","volume-title":"Zero-shot compositional concept learning. arXiv preprint arXiv:2107.05176","author":"Xu Guangyue","year":"2021","unstructured":"Guangyue Xu, Parisa Kordjamshidi, and Joyce Y Chai. 2021a. Zero-shot compositional concept learning. arXiv preprint arXiv:2107.05176 (2021)."},{"key":"e_1_3_2_1_31_1","volume-title":"Relation-aware Compositional Zero-shot Learning for Attribute-Object Pair Recognition","author":"Xu Ziwei","year":"2021","unstructured":"Ziwei Xu, Guangzhi Wang, Yongkang Wong, and Mohan S Kankanhalli. 2021b. Relation-aware Compositional Zero-shot Learning for Attribute-Object Pair Recognition. IEEE Transactions on Multimedia (2021)."},{"key":"e_1_3_2_1_32_1","volume-title":"Dual-stream contrastive learning for compositional zero-shot recognition","author":"Yang Yanhua","year":"2023","unstructured":"Yanhua Yang, Rui Pan, Xiangyu Li, Xu Yang, and Cheng Deng. 2023. Dual-stream contrastive learning for compositional zero-shot recognition. IEEE Transactions on Multimedia (2023)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.32"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.594"},{"key":"e_1_3_2_1_35_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Yuksekgonul Mert","year":"2022","unstructured":"Mert Yuksekgonul, Federico Bianchi, Pratyusha Kalluri, Dan Jurafsky, and James Zou. 2022. When and why vision-language models behave like bags-of-words, and what to do about it?. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01307"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00174"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755492","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:18:42Z","timestamp":1765307922000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755492"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":38,"alternative-id":["10.1145\/3746027.3755492","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755492","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}