{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:05:44Z","timestamp":1750309544409,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T00:00:00Z","timestamp":1745280000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"Natural Science Foundation of China","award":["Grant No. U21B2026, 62372260"],"award-info":[{"award-number":["Grant No. U21B2026, 62372260"]}]},{"name":"Quan Cheng Laboratory","award":["Grant No. QCLZD202301"],"award-info":[{"award-number":["Grant No. QCLZD202301"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,22]]},"DOI":"10.1145\/3696410.3714772","type":"proceedings-article","created":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T22:57:28Z","timestamp":1745362648000},"page":"293-303","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PerSRV: Personalized Sticker Retrieval with Vision-Language Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-5562-5878","authenticated-orcid":false,"given":"Heng Er Metilda","family":"Chee","sequence":"first","affiliation":[{"name":"DCST, Tsinghua University, Beijing, China and Quan Cheng Laboratory, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8875-1850","authenticated-orcid":false,"given":"Jiayin","family":"Wang","sequence":"additional","affiliation":[{"name":"DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9393-4854","authenticated-orcid":false,"given":"Zhiqiang","family":"Guo","sequence":"additional","affiliation":[{"name":"DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5604-7527","authenticated-orcid":false,"given":"Weizhi","family":"Ma","sequence":"additional","affiliation":[{"name":"AIR, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3158-1920","authenticated-orcid":false,"given":"Min","family":"Zhang","sequence":"additional","affiliation":[{"name":"DCST, Tsinghua University, Beijing, China and Quan Cheng Laboratory, Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,4,22]]},"reference":[{"key":"e_1_3_2_1_2_1","volume-title":"Oh (Eds.)","volume":"35","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, Antoine Miech, Iain Barr, Yana Hasson, Karel Lenc, Arthur Mensch, Katherine Millican, Malcolm Reynolds, Roman Ring, Eliza Rutherford, Serkan Cabi, Tengda Han, Zhitao Gong, Sina Samangooei, Marianne Monteiro, Jacob L Menick, Sebastian Borgeaud, Andy Brock, Aida Nematzadeh, Sahand Sharifzadeh, Miko\u0142 aj Bi\u0144kowski, Ricardo Barreira, Oriol Vinyals, Andrew Zisserman, and Kar\u00e9n Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., 23716--23736. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/960a172bc7fbf0177ccccbb411a7d800-Paper-Conference.pdf"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681522"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3301299"},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Dai Wenliang","year":"2024","unstructured":"Wenliang Dai, Junnan Li, Dongxu Li, Anthony Meng Huat Tiong, Junqi Zhao, Weisheng Wang, Boyang Li, Pascale Fung, and Steven Hoi. 2024. InstructBLIP: towards general-purpose vision-language models with instruction tuning. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS '23). Curran Associates Inc., Red Hook, NY, USA, Article 2142, 18 pages."},{"key":"e_1_3_2_1_7_1","volume-title":"PP-OCR: A Practical Ultra Lightweight OCR System. arxiv","author":"Du Yuning","year":"2009","unstructured":"Yuning Du, Chenxia Li, Ruoyu Guo, Xiaoting Yin, Weiwei Liu, Jun Zhou, Yifan Bai, Zilin Yu, Yehua Yang, Qingqing Dang, and Haoshuang Wang. 2020. PP-OCR: A Practical Ultra Lightweight OCR System. arxiv: 2009.09941 [cs.CV] https:\/\/arxiv.org\/abs\/2009.09941"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1207\/s15327744joce1603&4_2"},{"key":"e_1_3_2_1_9_1","volume-title":"Towards expressive communication with internet memes: A new multimodal conversation dataset and benchmark. arXiv preprint arXiv:2109.01839","author":"Fei Zhengcong","year":"2021","unstructured":"Zhengcong Fei, Zekang Li, Jinchao Zhang, Yang Feng, and Jie Zhou. 2021. Towards expressive communication with internet memes: A new multimodal conversation dataset and benchmark. arXiv preprint arXiv:2109.01839 (2021)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3652583.3657627"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380191"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3429980"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics. 6795--6804","author":"Ge Feng","year":"2022","unstructured":"Feng Ge, Weizhao Li, Haopeng Ren, and Yi Cai. 2022. Towards exploiting sticker for multimodal sentiment analysis in social media: A new dataset and baseline. In Proceedings of the 29th International Conference on Computational Linguistics. 6795--6804."},{"key":"e_1_3_2_1_14_1","volume-title":"ICWSM Workshops.","author":"Ge Jing","year":"2020","unstructured":"Jing Ge. 2020. The Anatomy of Memetic Stickers: An Analysis of Sticker Competition on Chinese Social Media.. In ICWSM Workshops."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967195"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403324"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580990"},{"key":"e_1_3_2_1_18_1","series-title":"Journal of physics: conference series","volume-title":"Laughing at one's self: A study of self-reflective internet memes","author":"Turhan Kariko Abdul Aziz","unstructured":"Abdul Aziz Turhan Kariko and Nonny Anasih. 2019. Laughing at one's self: A study of self-reflective internet memes. In Journal of physics: conference series, Vol. 1175. IOP Publishing, 012250."},{"key":"e_1_3_2_1_19_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Koh Jing Yu","year":"2024","unstructured":"Jing Yu Koh, Daniel Fried, and Russ R Salakhutdinov. 2024. Generating images with multimodal language models. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/3477.764879"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i08.7019"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3545862.3545889"},{"key":"e_1_3_2_1_23_1","volume-title":"International conference on machine learning. PMLR","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730--19742."},{"key":"e_1_3_2_1_24_1","volume-title":"International conference on machine learning. PMLR, 12888--12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In International conference on machine learning. PMLR, 12888--12900."},{"key":"e_1_3_2_1_25_1","unstructured":"Bin Liang Bingbing Wang Zhixin Bai Qiwei Lang Mingwei Sun Kaiheng Hou Lanjun Zhou Ruifeng Xu and Kam-Fai Wong. 2024. Reply with Sticker: New Dataset and Model for Sticker Retrieval. arxiv: 2403.05427 [cs.MM] https:\/\/arxiv.org\/abs\/2403.05427"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvcir.2011.01.005"},{"key":"e_1_3_2_1_27_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li and Yong Jae Lee. 2023a. Improved Baselines with Visual Instruction Tuning."},{"key":"e_1_3_2_1_28_1","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong Jae Lee. 2023b. Visual Instruction Tuning. In NeurIPS."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548407"},{"key":"e_1_3_2_1_30_1","unstructured":"Xu Ming. 2022. text2vec: A Tool for Text to Vector. https:\/\/github.com\/shibing624\/text2vec"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318."},{"key":"e_1_3_2_1_32_1","volume-title":"EmojiLM: Modeling the New Emoji Language. arXiv preprint arXiv:2311.01751","author":"Peng Letian","year":"2023","unstructured":"Letian Peng, Zilong Wang, Hang Liu, Zihan Wang, and Jingbo Shang. 2023. EmojiLM: Modeling the New Emoji Language. arXiv preprint arXiv:2311.01751 (2023)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680978"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2043932.2043957"},{"key":"e_1_3_2_1_36_1","volume-title":"A Survey of Deep Learning Approaches for OCR and Document Understanding. arxiv","author":"Subramani Nishant","year":"2011","unstructured":"Nishant Subramani, Alexandre Matton, Malcolm Greaves, and Adrian Lam. 2021. A Survey of Deep Learning Approaches for OCR and Document Understanding. arxiv: 2011.13534 [cs.CL] https:\/\/arxiv.org\/abs\/2011.13534"},{"key":"e_1_3_2_1_37_1","volume-title":"Telegram Sticker Maker. https:\/\/telegram.org\/blog\/sticker-maker\/fa'setln=en [Online","author":"Blog Telegram","year":"2025","unstructured":"Telegram Blog. 2024. Telegram Sticker Maker. https:\/\/telegram.org\/blog\/sticker-maker\/fa'setln=en [Online; accessed February 4, 2025]."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 22nd annual conference of the European Association for Machine Translation. 479--480","author":"Tiedemann J\u00f6rg","year":"2020","unstructured":"J\u00f6rg Tiedemann and Santhosh Thottingal. 2020. OPUS-MT--building open translation services for the world. In Proceedings of the 22nd annual conference of the European Association for Machine Translation. 479--480."},{"key":"e_1_3_2_1_39_1","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arxiv: 2302.13971 [cs.CL] https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_2_1_40_1","volume-title":"Towards Real-World Stickers Use: A New Dataset for Multi-Tag Sticker Recognition. arXiv preprint arXiv:2403.05428","author":"Wang Bingbing","year":"2024","unstructured":"Bingbing Wang, Bin Liang, Chun-Mei Feng, Wangmeng Zuo, Zhixin Bai, Shijue Huang, Kam-Fai Wong, Xi Zeng, and Ruifeng Xu. 2024. Towards Real-World Stickers Use: A New Dataset for Multi-Tag Sticker Recognition. arXiv preprint arXiv:2403.05428 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3343123"},{"key":"e_1_3_2_1_42_1","unstructured":"WeChat. 2024. 2024 National Information Retrieval Challenge Cup (CCIR Cup). https:\/\/algo.weixin.qq.com\/ Accessed: 2024--10-09."},{"key":"e_1_3_2_1_43_1","volume-title":"Using Stickers on WeChat. https:\/\/help.wechat.com\/cgi-bin\/micromsg-bin\/oshelpcenter?opcode=2&lang=en&plat=ios&id=1208117b2mai141024nu67FJ [Online","author":"WeChat Help Center","year":"2025","unstructured":"WeChat Help Center. 2024. Using Stickers on WeChat. https:\/\/help.wechat.com\/cgi-bin\/micromsg-bin\/oshelpcenter?opcode=2&lang=en&plat=ios&id=1208117b2mai141024nu67FJ [Online; accessed February 4, 2025]."},{"key":"e_1_3_2_1_44_1","volume-title":"Using Stickers on WhatsApp. https:\/\/faq.whatsapp.com\/639351827594474\/?helpref=uf_share [Online","author":"Support WhatsApp","year":"2025","unstructured":"WhatsApp Support. 2024. Using Stickers on WhatsApp. https:\/\/faq.whatsapp.com\/639351827594474\/?helpref=uf_share [Online; accessed February 4, 2025]."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680987"},{"key":"e_1_3_2_1_46_1","unstructured":"An Yang Junshu Pan Junyang Lin Rui Men Yichang Zhang Jingren Zhou and Chang Zhou. 2023. Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese. arxiv: 2211.01335 [cs.CV] https:\/\/arxiv.org\/abs\/2211.01335"},{"key":"e_1_3_2_1_47_1","volume-title":"StickerConv: Generating Multimodal Empathetic Responses from Scratch. arXiv preprint arXiv:2402.01679","author":"Zhang Yiqun","year":"2024","unstructured":"Yiqun Zhang, Fanheng Kong, Peidong Wang, Shuang Sun, Lingshuai Wang, Shi Feng, Daling Wang, Yifei Zhang, and Kaisong Song. 2024. StickerConv: Generating Multimodal Empathetic Responses from Scratch. arXiv preprint arXiv:2402.01679 (2024)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.241"},{"key":"e_1_3_2_1_49_1","volume-title":"Sticker820k: Empowering interactive retrieval with stickers. arXiv preprint arXiv:2306.06870","author":"Zhao Sijie","year":"2023","unstructured":"Sijie Zhao, Yixiao Ge, Zhongang Qi, Lin Song, Xiaohan Ding, Zehua Xie, and Ying Shan. 2023. Sticker820k: Empowering interactive retrieval with stickers. arXiv preprint arXiv:2306.06870 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"Minigpt-5: Interleaved vision-and-language generation via generative vokens. arXiv preprint arXiv:2310.02239","author":"Zheng Kaizhi","year":"2023","unstructured":"Kaizhi Zheng, Xuehai He, and Xin Eric Wang. 2023. Minigpt-5: Interleaved vision-and-language generation via generative vokens. arXiv preprint arXiv:2310.02239 (2023)."},{"key":"e_1_3_2_1_51_1","volume-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)."}],"event":{"name":"WWW '25: The ACM Web Conference 2025","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"],"location":"Sydney NSW Australia","acronym":"WWW '25"},"container-title":["Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714772","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696410.3714772","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:41Z","timestamp":1750295921000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714772"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,22]]},"references-count":50,"alternative-id":["10.1145\/3696410.3714772","10.1145\/3696410"],"URL":"https:\/\/doi.org\/10.1145\/3696410.3714772","relation":{},"subject":[],"published":{"date-parts":[[2025,4,22]]},"assertion":[{"value":"2025-04-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}