{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T11:29:55Z","timestamp":1764588595133,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3696409.3700286","type":"proceedings-article","created":{"date-parts":[[2024,12,28]],"date-time":"2024-12-28T09:55:23Z","timestamp":1735379723000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["CSCCap: Plugging Sparse Coding in Zero-Shot Image Captioning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8090-0938","authenticated-orcid":false,"given":"Yu","family":"Song","sequence":"first","affiliation":[{"name":"Henan University, Kaifeng, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4517-267X","authenticated-orcid":false,"given":"Xiaohui","family":"Yang","sequence":"additional","affiliation":[{"name":"Henan University, Kaifeng, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8799-1478","authenticated-orcid":false,"given":"Rongping","family":"Huang","sequence":"additional","affiliation":[{"name":"Southern University of Science and Technology, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9129-7344","authenticated-orcid":false,"given":"Bai","family":"Haifeng","sequence":"additional","affiliation":[{"name":"DEEPEXI, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7070-9723","authenticated-orcid":false,"given":"Lili","family":"Yang","sequence":"additional","affiliation":[{"name":"Southern University of Science and Technology, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,28]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Aviad Aberdam Jeremias Sulam and Michael Elad. 2019. Multi-layer sparse coding: The holistic way. SIAM Journal on Mathematics of Data Science 1 1 (2019) 46\u201377.","DOI":"10.1137\/18M1183352"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.150"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.57"},{"key":"e_1_3_3_1_6_2","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared\u00a0D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et\u00a0al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877\u20131901."},{"key":"e_1_3_3_1_7_2","unstructured":"Xinlei Chen Hao Fang Tsung-Yi Lin Ramakrishna Vedantam Saurabh Gupta Piotr Doll\u00e1r and C\u00a0Lawrence Zitnick. 2015. Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1504.00325 (2015)."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-3348"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00291"},{"key":"e_1_3_3_1_11_2","first-page":"4904","volume-title":"International conference on machine learning","author":"Jia Chao","year":"2021","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2021. Scaling up visual and vision-language representation learning with noisy text supervision. In International conference on machine learning. PMLR, 4904\u20134916."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Kenneth Kreutz-Delgado Joseph\u00a0F Murray Bhaskar\u00a0D Rao Kjersti Engan Te-Won Lee and Terrence\u00a0J Sejnowski. 2003. Dictionary learning algorithms for sparse representation. Neural computation 15 2 (2003) 349\u2013396.","DOI":"10.1162\/089976603762552951"},{"key":"e_1_3_3_1_13_2","unstructured":"Wei Li Linchao Zhu Longyin Wen and Yi Yang. 2023. Decap: Decoding clip latents for zero-shot captioning via text-only training. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.03032 (2023)."},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"David Nukrai Ron Mokady and Amir Globerson. 2022. Text-only training for image captioning using noise-injected clip. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.00575 (2022).","DOI":"10.18653\/v1\/2022.findings-emnlp.299"},{"key":"e_1_3_3_1_16_2","first-page":"311","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318. https:\/\/dl.acm.org\/doi\/10.3115\/1073083.1073135"},{"key":"e_1_3_3_1_17_2","unstructured":"Vardan Papyan Yaniv Romano and Michael Elad. 2017. Convolutional neural networks analyzed via convolutional sparse coding. Journal of Machine Learning Research 18 83 (2017) 1\u201352."},{"key":"e_1_3_3_1_18_2","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_1_19_2","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et\u00a0al. 2019. Language models are unsupervised multitask learners. OpenAI blog 1 8 (2019) 9."},{"key":"e_1_3_3_1_20_2","unstructured":"Yixuan Su Tian Lan Yahui Liu Fangyu Liu Dani Yogatama Yan Wang Lingpeng Kong and Nigel Collier. 2022. Language models can see: Plugging visual controls in text generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.02655 (2022)."},{"key":"e_1_3_3_1_21_2","unstructured":"Jeremias Sulam Vardan Papyan Yaniv Romano and Michael Elad. 2018. Multilayer convolutional sparse modeling: Pursuit and dictionary learning. IEEE Transactions on Signal Processing 66 15 (2018) 4090\u20134104."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Ju Sun Qing Qu and John Wright. 2016. Complete dictionary recovery over the sphere I: Overview and the geometric picture. IEEE Transactions on Information Theory 63 2 (2016) 853\u2013884.","DOI":"10.1109\/TIT.2016.2632162"},{"key":"e_1_3_3_1_23_2","unstructured":"Yoad Tewel Yoav Shalev Idan Schwartz and Lior Wolf. 2021. Zero-shot image-to-text generation for visual-semantic arithmetic. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.14447 2 (2021)."},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854992"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"Peter Young Alice Lai Micah Hodosh and Julia Hockenmaier. 2014. From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Transactions of the Association for Computational Linguistics 2 (2014) 67\u201378.","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_3_1_27_2","unstructured":"Andy Zeng Maria Attarian Brian Ichter Krzysztof Choromanski Adrian Wong Stefan Welker Federico Tombari Aveek Purohit Michael Ryoo Vikas Sindhwani et\u00a0al. 2022. Socratic models: Composing zero-shot multimodal reasoning with language. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.00598 (2022)."},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Zhiyang Zhang and Shihua Zhang. 2021. Towards understanding residual and dilated dense neural networks via convolutional sparse coding. National science review 8 3 (2021) nwaa159.","DOI":"10.1093\/nsr\/nwaa159"}],"event":{"name":"MMAsia '24: ACM Multimedia Asia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Auckland New Zealand","acronym":"MMAsia '24"},"container-title":["Proceedings of the 6th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700286","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696409.3700286","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:16Z","timestamp":1750295416000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700286"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":27,"alternative-id":["10.1145\/3696409.3700286","10.1145\/3696409"],"URL":"https:\/\/doi.org\/10.1145\/3696409.3700286","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}