{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T08:51:19Z","timestamp":1777107079528,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":18,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,14]]},"DOI":"10.1145\/3787279.3788494","type":"proceedings-article","created":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T07:38:47Z","timestamp":1777102727000},"page":"197-202","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Category-Level Segmentation of Repeated Objects using SAM2 Memory System"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2005-0986","authenticated-orcid":false,"given":"Jiayi","family":"Wang","sequence":"first","affiliation":[{"name":"Corporate Research, Robert Bosch GmbH, Renningen, Germany and Pattern Recognition Lab, Friedrich-Alexander-Universit, Erlangen, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5228-3397","authenticated-orcid":false,"given":"Matthias","family":"Kayser","sequence":"additional","affiliation":[{"name":"Corporate Research, Robert Bosch GmbH, Renningen, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8587-8894","authenticated-orcid":false,"given":"Yipeng","family":"Sun","sequence":"additional","affiliation":[{"name":"Pattern Recognition Lab, Friedrich-Alexander-Universit, Erlangen, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,25]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298799"},{"key":"e_1_3_3_1_3_2","unstructured":"Tom\u00a0B. Brown and et al.2020. Language Models Are Few-Shot Learners. Advances in Neural Information Processing Systems (NeurIPS) 2020 33 (2020) 1877\u20131901."},{"key":"e_1_3_3_1_4_2","volume-title":"Conference on Neural Information Processing Systems (NeurIPS), 2021","author":"Cheng Bohan","year":"2021","unstructured":"Bohan Cheng, Alexander Schwing, and Alexander Kirillov. 2021. Per-Pixel Classification is Not All You Need for Semantic Segmentation. In Conference on Neural Information Processing Systems (NeurIPS), 2021."},{"key":"e_1_3_3_1_5_2","first-page":"4171","volume-title":"Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT), 2019","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT), 2019. 4171\u20134186."},{"key":"e_1_3_3_1_6_2","unstructured":"Khronos Group. 2011. OpenMax: The Open Multimedia API. https:\/\/www.khronos.org\/omx. Version 1.2 accessed 13-Feb-2025."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_3_1_8_2","first-page":"5830","volume-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2020","author":"Joseph Karan","year":"2020","unstructured":"Karan Joseph, Jayakrishna Rajasegaran, Salman Khan, Fahad\u00a0Shahbaz Khan, and Vineeth\u00a0N. Balasubramanian. 2020. Open-World Object Detection. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2020. 5830\u20135840."},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.00371"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2017.13"},{"key":"e_1_3_3_1_11_2","unstructured":"Shu Liu Zelin Zeng Tao Ren Feng Li Hong Zhang Jian Yang Chen Li Jun Yang Hong Su Jie Zhu and Li Zhang. 2023. Grounding DINO: Marrying DINO with Grounded Pre-training for Open-set Object Detection. CoRR 2023 (2023). https:\/\/arxiv.org\/abs\/2303.05499"},{"key":"e_1_3_3_1_12_2","unstructured":"Z. Long G. Killick R. McCreadie and G.\u00a0A. Camarasa. 2024. RoboLLM: Robotic Vision Tasks Grounded on Multimodal Large Language Models. https:\/\/arxiv.org\/abs\/2310.10221. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.10221."},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"David\u00a0G. Lowe. 2004. Distinctive Image Features from Scale-Invariant Keypoints. International Journal of Computer Vision 2004 60 (2004) 91\u2013110.","DOI":"10.1023\/B:VISI.0000029664.99615.94"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Chaitanya Mitash Fangzhou Wang Shalini Lu Vishal Terhuja T. Garaas F. Polido and M. Nambi. 2023. ArmBench: An Object-Centric Benchmark Dataset for Robotic Manipulation. (2023).","DOI":"10.1109\/ICRA48891.2023.10160846"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_38"},{"key":"e_1_3_3_1_16_2","unstructured":"Nikhila Ravi Vincent Gabeur Yuan-Ting Hu Ruoxi Hu Chaitanya Ryali Tianhao Ma Hossam Khedr Rainer R\u00e4dle Chen Rolland Lucas Gustafson Evan Mintun Jianan Pan Kiran\u00a0V. Alwala Nicolas Carion Chih-Yao Wu Ross Girshick Piotr Doll\u00e1r and Christoph Feichtenhofer. 2024. SAM 2: Segment Anything in Images and Videos. (2024). arxiv:https:\/\/arXiv.org\/abs\/2408.00714\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2408.00714"},{"key":"e_1_3_3_1_17_2","unstructured":"Tao Ren Shu Liu Andrew Zeng Jiawei Lin Kun Li Hao Cao Jie Chen Xia Huang Yifan Chen Feng Yan Zelin Zeng Hong Zhang Feng Li Jian Yang Hong Li Qi Jiang and Li Zhang. 2024. Grounded SAM: Assembling Open-World Models for Diverse Visual Tasks. (2024). arxiv:https:\/\/arXiv.org\/abs\/2401.14159\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2401.14159"},{"key":"e_1_3_3_1_18_2","volume-title":"Conference on Neural Information Processing Systems (NeurIPS), 2020","author":"Tancik Michael","year":"2020","unstructured":"Michael Tancik, Pratul Srinivasan, Ben Mildenhall, S. Fridovich-Keil, N. Raghavan, Utkarsh Singhal, Ravi Ramamoorthi, Jonathan Barron, and Ren Ng. 2020. Fourier Features Let Networks Learn High Frequency Functions in Low Dimensional Domains. In Conference on Neural Information Processing Systems (NeurIPS), 2020."},{"key":"e_1_3_3_1_19_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All You Need. arXiv preprint 2017 arXiv:1706.03762 (2017). https:\/\/arxiv.org\/pdf\/1706.03762v5"}],"event":{"name":"ICAAI 2025: 2025 9th International Conference on Advances in Artificial Intelligence","location":"Manchester United Kingdom","acronym":"ICAAI 2025"},"container-title":["Proceedings of the 2025 9th International Conference on Advances in Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3787279.3788494","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T08:23:15Z","timestamp":1777105395000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3787279.3788494"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,14]]},"references-count":18,"alternative-id":["10.1145\/3787279.3788494","10.1145\/3787279"],"URL":"https:\/\/doi.org\/10.1145\/3787279.3788494","relation":{},"subject":[],"published":{"date-parts":[[2025,11,14]]},"assertion":[{"value":"2026-04-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}