{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T08:42:15Z","timestamp":1774428135052,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755426","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"1764-1773","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["TrueCount: Improving Open-World Object Counting with Visual-Language Models and Dynamic Multi-Modal Inputs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3105-6213","authenticated-orcid":false,"given":"Ziqiang","family":"Shi","sequence":"first","affiliation":[{"name":"Fujitsu Research &amp; Development Center Co.,LTD., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2417-7115","authenticated-orcid":false,"given":"Rujie","family":"Liu","sequence":"additional","affiliation":[{"name":"Fujitsu Research &amp; Development Center Co.,LTD., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2710-2587","authenticated-orcid":false,"given":"Jun","family":"Takahashi","sequence":"additional","affiliation":[{"name":"Fujitsu Limited, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4223-5857","authenticated-orcid":false,"given":"Shan","family":"Jiang","sequence":"additional","affiliation":[{"name":"Fujitsu Limited, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1117\/1.JMI.6.1.014006"},{"key":"e_1_3_2_1_2_1","volume-title":"Open-world text-specified object counting. arXiv preprint arXiv:2306.01851","author":"Amini-Naieni Niki","year":"2023","unstructured":"Niki Amini-Naieni, Kiana Amini-Naieni, Tengda Han, and Andrew Zisserman. 2023. Open-world text-specified object counting. arXiv preprint arXiv:2306.01851 (2023)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"N. Amini-Naieni T. Han and A. Zisserman. 2024. CountGD: Multi-Modal Open-World Counting. In Advances in Neural Information Processing Systems (NeurIPS).","DOI":"10.52202\/079017-1547"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5946681"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.236"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00096"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01902"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00625"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01607"},{"key":"e_1_3_2_1_10_1","first-page":"4171","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171-4186."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01169"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01615"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611789"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00476"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2018.2850149"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.28050"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00401"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_38"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01901"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"e_1_3_2_1_21_1","volume-title":"Countr: Transformer-based generalised visual counting. arXiv preprint arXiv:2208.13721","author":"Liu Chang","year":"2022","unstructured":"Chang Liu, Yujie Zhong, Andrew Zisserman, and Weidi Xie. 2022. Countr: Transformer-based generalised visual counting. arXiv preprint arXiv:2208.13721 (2022)."},{"key":"e_1_3_2_1_22_1","volume-title":"European Conference on Computer Vision. Springer, 38-55","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Qing Jiang, Chunyuan Li, Jianwei Yang, Hang Su, et al., 2024. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In European Conference on Computer Vision. Springer, 38-55."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_24_1","volume-title":"CLIP-EBC: CLIP Can Count Accurately through Enhanced Blockwise Classification. arXiv","author":"Ma Y","year":"2024","unstructured":"Y Ma, V Sanchez, and T Guha. [n.d.]. CLIP-EBC: CLIP Can Count Accurately through Enhanced Blockwise Classification. arXiv 2024. arXiv preprint arXiv:2403.09281 ( [n.,d.])."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46487-9_48"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00294"},{"key":"e_1_3_2_1_27_1","volume-title":"A Novel Unified Architecture for Low-Shot Counting by Detection and Segmentation. arXiv preprint arXiv:2409.18686","author":"Pelhan Jer","year":"2024","unstructured":"Jer Pelhan, Alan Luke\u017ei\u010d, Vitjan Zavrtanik, and Matej Kristan. 2024a. A Novel Unified Architecture for Low-Shot Counting by Detection and Segmentation. arXiv preprint arXiv:2409.18686 (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02198"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00399"},{"key":"e_1_3_2_1_30_1","volume-title":"International conference on machine learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748-8763."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00340"},{"key":"e_1_3_2_1_32_1","unstructured":"Nikhila Ravi Valentin Gabeur Yuan-Ting Hu Ronghang Hu Chaitanya Ryali Tengyu Ma Haitham Khedr Roman R\u00e4dle Chloe Rolland Laura Gustafson et al. 2024. Sam 2: Segment anything in images and videos. arXiv preprint arXiv:2408.00714 (2024)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00549"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01551"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00039"},{"key":"e_1_3_2_1_36_1","first-page":"523","volume-title":"Lima","author":"Shirokikh Boris","year":"2020","unstructured":"Boris Shirokikh, Alexey Shevtsov, Anvar Kurmukov, Alexandra Dalechina, Egor Krivov, Valery Kostjuchenko, Andrey Golanov, and Mikhail Belyaev. 2020. Universal loss reweighting to balance lesion size inequality in 3D medical image segmentation. In Medical Image Computing and Computer Assisted Intervention-MICCAI 2020: 23rd International Conference, Lima, Peru, October 4-8, 2020, Proceedings, Part IV 23. Springer, 523-532."},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 18872-18881","author":"Juki\u0107 Nikola","year":"2023","unstructured":"Nikola DJuki\u0107, Alan Luke\u017ei\u010d, Vitjan Zavrtanik, and Matej Kristan. 2023. A low-shot object counting network with iterative prototype adaptation. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 18872-18881."},{"key":"e_1_3_2_1_38_1","unstructured":"Ao Wang Hui Chen Lihao Liu Kai Chen Zijia Lin Jungong Han and Guiguang Ding. 2024. YOLOv10: Real-Time End-to-End Object Detection. Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01492"},{"key":"e_1_3_2_1_40_1","volume-title":"Zero-shot object counting with language-vision models. arXiv preprint arXiv:2309.13097","author":"Xu Jingyi","year":"2023","unstructured":"Jingyi Xu, Hieu Le, and Dimitris Samaras. 2023a. Zero-shot object counting with language-vision models. arXiv preprint arXiv:2309.13097 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"Global guidance network for breast lesion segmentation in ultrasound images. Medical image analysis","author":"Xue Cheng","year":"2021","unstructured":"Cheng Xue, Lei Zhu, Huazhu Fu, Xiaowei Hu, Xiaomeng Li, Hai Zhang, and Pheng-Ann Heng. 2021. Global guidance network for breast lesion segmentation in ultrasound images. Medical image analysis, Vol. 70 (2021), 101989."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01055"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01330"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00840"},{"key":"e_1_3_2_1_45_1","first-page":"21875","article-title":"Depth anything v2","volume":"37","author":"Yang Lihe","year":"2024","unstructured":"Lihe Yang, Bingyi Kang, Zilong Huang, Zhen Zhao, Xiaogang Xu, Jiashi Feng, and Hengshuang Zhao. 2024. Depth anything v2. Advances in Neural Information Processing Systems, Vol. 37 (2024), 21875-21911.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00840"},{"key":"e_1_3_2_1_47_1","volume-title":"European Conference on Computer Vision. Springer, 368-385","author":"Zhu Huilin","year":"2024","unstructured":"Huilin Zhu, Jingling Yuan, Zhengwei Yang, Yu Guo, Zheng Wang, Xian Zhong, and Shengfeng He. 2024. Zero-shot object counting with good exemplars. In European Conference on Computer Vision. Springer, 368-385."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755426","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:12:32Z","timestamp":1765339952000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755426"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":47,"alternative-id":["10.1145\/3746027.3755426","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755426","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}