{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T06:39:16Z","timestamp":1743143956496,"version":"3.40.3"},"publisher-location":"Cham","reference-count":82,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031726729"},{"type":"electronic","value":"9783031726736"}],"license":[{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T00:00:00Z","timestamp":1729555200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72673-6_12","type":"book-chapter","created":{"date-parts":[[2024,10,21]],"date-time":"2024-10-21T16:03:50Z","timestamp":1729526630000},"page":"211-230","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["R$$^2$$-Bench: Benchmarking the\u00a0Robustness of\u00a0Referring Perception Models Under Perturbations"],"prefix":"10.1007","author":[{"given":"Xiang","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinglu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaohao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rita","family":"Singh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kashu","family":"Yamazaki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaonan","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bhiksha","family":"Raj","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,22]]},"reference":[{"key":"12_CR1","unstructured":"Achiam, J., et al.: GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Ahn, H., et al.: Visually grounding language instruction for history-dependent manipulation. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 675\u2013682. IEEE (2022)","DOI":"10.1109\/ICRA46639.2022.9812279"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Look, listen and learn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 609\u2013617 (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., Zisserman, A.: Objects that sound. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 435\u2013451 (2018)","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Botach, A., Zheltonozhskii, E., Baskin, C.: End-to-end referring video object segmentation with multimodal transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4985\u20134995 (2022)","DOI":"10.1109\/CVPR52688.2022.00493"},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Chen, F., Zhang, H., Hu, K., Huang, Y.K., Zhu, C., Savvides, M.: Enhanced training of query-based object detection via selective query recollection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23756\u201323765 (2023)","DOI":"10.1109\/CVPR52729.2023.02275"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Chen, F., et al.: Unitail: detecting, reading, and matching in retail scene (2022). https:\/\/arxiv.org\/abs\/2204.00298","DOI":"10.1007\/978-3-031-20071-7_41"},{"key":"12_CR8","unstructured":"Chen, F., Zhang, H., Yang, Z., Chen, H., Hu, K., Savvides, M.: Rtgen: generating region-text pairs for open-vocabulary object detection (2024). https:\/\/arxiv.org\/abs\/2405.19854"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Oh, S.W., Price, B., Lee, J.Y., Schwing, A.: Putting the object back into video object segmentation. arXiv preprint arXiv:2310.12982 (2023)","DOI":"10.1109\/CVPR52733.2024.00304"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Oh, S.W., Price, B., Schwing, A., Lee, J.Y.: Tracking anything with decoupled video segmentation. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00127"},{"key":"12_CR11","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1007\/978-3-031-19815-1_37","volume-title":"ECCV 2022","author":"HK Cheng","year":"2022","unstructured":"Cheng, H.K., Schwing, A.G.: XMem: long-term video object segmentation with an Atkinson-Shiffrin memory model. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13688, pp. 640\u2013658. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_37"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Wang, R., Pan, Z., Feng, R., Zhang, Y.: Look, listen, and attend: co-attention network for self-supervised audio-visual representation learning. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 3884\u20133892 (2020)","DOI":"10.1145\/3394171.3413869"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: Scannet: richly-annotated 3D reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5828\u20135839 (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Ding, H., Liu, C., He, S., Jiang, X., Loy, C.C.: Mevis: a large-scale benchmark for video segmentation with motion expressions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2694\u20132703 (2023)","DOI":"10.1109\/ICCV51070.2023.00254"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Ding, H., Liu, C., Wang, S., Jiang, X.: Vision-language transformer and query generation for referring segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16321\u201316330 (2021)","DOI":"10.1109\/ICCV48922.2021.01601"},{"key":"12_CR16","unstructured":"Du, Y., Li, S., Torralba, A., Tenenbaum, J.B., Mordatch, I.: Improving factuality and reasoning in language models through multiagent debate. arXiv preprint arXiv:2305.14325 (2023)"},{"key":"12_CR17","unstructured":"Gao, S., Chen, Z., Chen, G., Wang, W., Lu, T.: Avsegformer: audio-visual segmentation with transformer. arXiv preprint arXiv:2307.01146 (2023)"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Han, M., Wang, Y., Li, Z., Yao, L., Chang, X., Qiao, Y.: HTML: hybrid temporal-scale multimodal learning framework for referring video object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13414\u201313423 (2023)","DOI":"10.1109\/ICCV51070.2023.01234"},{"key":"12_CR19","doi-asserted-by":"crossref","unstructured":"Handa, A., Whelan, T., McDonald, J., Davison, A.: A benchmark for RGB-D visual odometry, 3D reconstruction and SLAM. In: IEEE International Conference on Robotics and Automation, ICRA, Hong Kong, China (2014)","DOI":"10.1109\/ICRA.2014.6907054"},{"key":"12_CR20","unstructured":"Hendrycks, D., Dietterich, T.: Benchmarking neural network robustness to common corruptions and perturbations. arXiv preprint arXiv:1903.12261 (2019)"},{"key":"12_CR21","unstructured":"Hu, Y., Lin, F., Zhang, T., Yi, L., Gao, Y.: Look before you leap: unveiling the power of GPT-4V in robotic vision-language planning. arXiv preprint arXiv:2311.17842 (2023)"},{"key":"12_CR22","unstructured":"Huang, W., et al.: Grounded decoding: guiding text generation with grounded models for robot control. arXiv preprint arXiv:2303.00855 (2023)"},{"key":"12_CR23","doi-asserted-by":"crossref","unstructured":"Jatavallabhula, K.M., et al.: Conceptfusion: open-set multimodal 3D mapping. arXiv preprint arXiv:2302.07241 (2023)","DOI":"10.15607\/RSS.2023.XIX.066"},{"key":"12_CR24","unstructured":"Ke, L., et al.: Segment anything in high quality. In: NeurIPS (2023)"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Khoreva, A., Rohrbach, A., Schiele, B.: Video object segmentation with language referring expressions. In: ACCV (2018)","DOI":"10.1007\/978-3-030-20870-7_8"},{"key":"12_CR26","unstructured":"Kirillov, A., et al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"12_CR27","doi-asserted-by":"crossref","unstructured":"Li, K., Yang, Z., Chen, L., Yang, Y., Xun, J.: CATR: combinatorial-dependence audio-queried transformer for audio-visual video segmentation. arXiv preprint arXiv:2309.09709 (2023)","DOI":"10.1145\/3581783.3611724"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Li, M., Li, S., Zhang, X., Zhang, L.: Univs: unified and universal video segmentation with prompts as queries. arXiv preprint arXiv:2402.18115 (2024)","DOI":"10.1109\/CVPR52733.2024.00311"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Li, X., Cao, H., Zhao, S., Li, J., Zhang, L., Raj, B.: Panoramic video salient object detection with ambisonic audio guidance. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 1424\u20131432 (2023)","DOI":"10.1609\/aaai.v37i2.25227"},{"key":"12_CR30","unstructured":"Li, X., Lin, C.C., Chen, Y., Liu, Z., Wang, J., Singh, R., Raj, B.: Paintseg: painting pixels for training-free segmentation. In: Thirty-Seventh Conference on Neural Information Processing Systems (2023)"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Li, X., Wang, J., Li, X., Lu, Y.: Hybrid instance-aware temporal fusion for online video instance segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 1429\u20131437 (2022)","DOI":"10.1609\/aaai.v36i2.20032"},{"key":"12_CR32","doi-asserted-by":"publisher","first-page":"7469","DOI":"10.1109\/TMM.2022.3222643","volume":"25","author":"X Li","year":"2022","unstructured":"Li, X., Wang, J., Li, X., Lu, Y.: Video instance segmentation by instance flow assembly. IEEE Trans. Multimedia 25, 7469\u20137479 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"12_CR33","doi-asserted-by":"crossref","unstructured":"Li, X., Wang, J., Xu, X., Li, X., Raj, B., Lu, Y.: Robust referring video object segmentation with cyclic structural consensus. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 22236\u201322245 (2023)","DOI":"10.1109\/ICCV51070.2023.02032"},{"key":"12_CR34","doi-asserted-by":"crossref","unstructured":"Li, X., et al.: Towards robust audiovisual segmentation in complex environments with quantization-based semantic decomposition. arXiv preprint arXiv:2310.00132 (2023)","DOI":"10.1109\/CVPR52733.2024.00327"},{"key":"12_CR35","doi-asserted-by":"crossref","unstructured":"Li, X., et al.: Towards noise-tolerant speech-referring video object segmentation: bridging speech and text. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 2283\u20132296 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.140"},{"key":"12_CR36","doi-asserted-by":"crossref","unstructured":"Liang, T., et al.: Encouraging divergent thinking in large language models through multi-agent debate. arXiv preprint arXiv:2305.19118 (2023)","DOI":"10.18653\/v1\/2024.emnlp-main.992"},{"key":"12_CR37","doi-asserted-by":"crossref","unstructured":"Liu, J., et al.: Polyformer: referring image segmentation as sequential polygon generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18653\u201318663 (2023)","DOI":"10.1109\/CVPR52729.2023.01789"},{"key":"12_CR38","doi-asserted-by":"crossref","unstructured":"Liu, J., Wang, Y., Ju, C., Ma, C., Zhang, Y., Xie, W.: Annotation-free audio-visual segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5604\u20135614 (2024)","DOI":"10.1109\/WACV57701.2024.00551"},{"key":"12_CR39","doi-asserted-by":"crossref","unstructured":"Liu, S., et al.: Grounding dino: marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"12_CR40","doi-asserted-by":"crossref","unstructured":"Liu, S., et al.: Dragon: a dialogue-based robot for assistive navigation with visual language grounding. IEEE Robot. Autom. Lett. (2024)","DOI":"10.1109\/LRA.2024.3362591"},{"key":"12_CR41","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., Camburu, O., Yuille, A.L., Murphy, K.: Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11\u201320 (2016)","DOI":"10.1109\/CVPR.2016.9"},{"key":"12_CR42","doi-asserted-by":"crossref","unstructured":"Miao, B., Bennamoun, M., Gao, Y., Mian, A.: Spectrum-guided multi-granularity referring video object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 920\u2013930 (2023)","DOI":"10.1109\/ICCV51070.2023.00091"},{"key":"12_CR43","unstructured":"Mo, S., Morgado, P.: A closer look at weakly-supervised audio-visual source localization. arXiv preprint arXiv:2209.09634 (2022)"},{"key":"12_CR44","doi-asserted-by":"crossref","unstructured":"Pan, W., et al.: Wnet: audio-guided video object segmentation via wavelet-based cross-modal denoising networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1320\u20131331 (2022)","DOI":"10.1109\/CVPR52688.2022.00138"},{"key":"12_CR45","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Van\u00a0Gool, L.: The 2017 davis challenge on video object segmentation. arXiv preprint arXiv:1704.00675 (2017)"},{"key":"12_CR46","doi-asserted-by":"crossref","unstructured":"Senocak, A., Oh, T.H., Kim, J., Yang, M.H., Kweon, I.S.: Learning to localize sound source in visual scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4358\u20134366 (2018)","DOI":"10.1109\/CVPR.2018.00458"},{"key":"12_CR47","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"208","DOI":"10.1007\/978-3-030-58555-6_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Seo","year":"2020","unstructured":"Seo, S., Lee, J.-Y., Han, B.: URVOS: unified referring video object segmentation network with a large-scale benchmark. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12360, pp. 208\u2013223. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58555-6_13"},{"issue":"2","key":"12_CR48","doi-asserted-by":"publisher","first-page":"4924","DOI":"10.1109\/LRA.2022.3150855","volume":"7","author":"J Sun","year":"2022","unstructured":"Sun, J., Huang, D.A., Lu, B., Liu, Y.H., Zhou, B., Garg, A.: Plate: visually-grounded planning with transformers in procedural tasks. IEEE Robot. Autom. Lett. 7(2), 4924\u20134930 (2022)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"12_CR49","doi-asserted-by":"crossref","unstructured":"Tang, J., Zheng, G., Yang, S.: Temporal collection and distribution for referring video object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15466\u201315476 (2023)","DOI":"10.1109\/ICCV51070.2023.01418"},{"key":"12_CR50","unstructured":"Team, G., et al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"12_CR51","doi-asserted-by":"crossref","unstructured":"Tziafas, G., Kasaei, H.: Few-shot visual grounding for natural human-robot interaction. In: 2021 IEEE International Conference on Autonomous Robot Systems and Competitions (ICARSC), pp. 50\u201355. IEEE (2021)","DOI":"10.1109\/ICARSC52212.2021.9429801"},{"key":"12_CR52","unstructured":"Wang, Z., Cai, S., Chen, G., Liu, A., Ma, X.S., Liang, Y.: Describe, explain, plan and select: interactive planning with LLMs enables open-world multi-task agents. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"12_CR53","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. In: Advances in Neural Information Processing Systems, vol. 35, pp. 24824\u201324837 (2022)"},{"key":"12_CR54","doi-asserted-by":"crossref","unstructured":"Wu, D., Wang, T., Zhang, Y., Zhang, X., Shen, J.: Onlinerefer: a simple online baseline for referring video object segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2761\u20132770 (2023)","DOI":"10.1109\/ICCV51070.2023.00259"},{"key":"12_CR55","doi-asserted-by":"crossref","unstructured":"Wu, J., Jiang, Y., Sun, P., Yuan, Z., Luo, P.: Language as queries for referring video object segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4974\u20134984 (2022)","DOI":"10.1109\/CVPR52688.2022.00492"},{"key":"12_CR56","doi-asserted-by":"crossref","unstructured":"Wu, J., Jiang, Y., Yan, B., Lu, H., Yuan, Z., Luo, P.: Segment every reference object in spatial and temporal spaces. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2538\u20132550 (2023)","DOI":"10.1109\/ICCV51070.2023.00240"},{"key":"12_CR57","unstructured":"Xi, Z., et al.: The rise and potential of large language model based agents: a survey. arXiv preprint arXiv:2309.07864 (2023)"},{"key":"12_CR58","doi-asserted-by":"crossref","unstructured":"Xiong, Y., et al.: Efficientsam: leveraged masked image pretraining for efficient segment anything. arXiv preprint arXiv:2312.00863 (2023)","DOI":"10.1109\/CVPR52733.2024.01525"},{"key":"12_CR59","unstructured":"Xu, N., et al.: Youtube-VOS: a large-scale video object segmentation benchmark. arXiv preprint arXiv:1809.03327 (2018)"},{"key":"12_CR60","doi-asserted-by":"crossref","unstructured":"Xu, X., Wang, J., Li, X., Lu, Y.: Reliable propagation-correction modulation for video object segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 2946\u20132954 (2022)","DOI":"10.1609\/aaai.v36i3.20200"},{"key":"12_CR61","doi-asserted-by":"crossref","unstructured":"Xu, X., Wang, J., Ming, X., Lu, Y.: Towards robust video object segmentation with adaptive object calibration. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 2709\u20132718 (2022)","DOI":"10.1145\/3503161.3547824"},{"key":"12_CR62","unstructured":"Xu, X., et al.: Customizable perturbation synthesis for robust slam benchmarking. arXiv preprint arXiv:2402.08125 (2024)"},{"key":"12_CR63","doi-asserted-by":"crossref","unstructured":"Xu, Z., Chen, Z., Zhang, Y., Song, Y., Wan, X., Li, G.: Bridging vision and language encoders: parameter-efficient tuning for referring image segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17503\u201317512 (2023)","DOI":"10.1109\/ICCV51070.2023.01605"},{"key":"12_CR64","doi-asserted-by":"crossref","unstructured":"Yamazaki, K., et al.: Open-fusion: real-time open-vocabulary 3D mapping and queryable scene representation. arXiv preprint arXiv:2310.03923 (2023)","DOI":"10.1109\/ICRA57147.2024.10610193"},{"key":"12_CR65","doi-asserted-by":"crossref","unstructured":"Yamazaki, K., et al.: Vlcap: vision-language with contrastive learning for coherent video paragraph captioning. In: 2022 IEEE International Conference on Image Processing (ICIP), pp. 3656\u20133661. IEEE (2022)","DOI":"10.1109\/ICIP46576.2022.9897766"},{"key":"12_CR66","doi-asserted-by":"crossref","unstructured":"Yamazaki, K., Vo, K., Truong, Q.S., Raj, B., Le, N.: Vltint: visual-linguistic transformer-in-transformer for coherent video paragraph captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 3081\u20133090 (2023)","DOI":"10.1609\/aaai.v37i3.25412"},{"key":"12_CR67","unstructured":"Yang, J., Gao, M., Li, Z., Gao, S., Wang, F., Zheng, F.: Track anything: segment anything meets videos (2023)"},{"key":"12_CR68","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, J., Tang, Y., Chen, K., Zhao, H., Torr, P.H.: LAVT: language-aware vision transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18155\u201318165 (2022)","DOI":"10.1109\/CVPR52688.2022.01762"},{"key":"12_CR69","unstructured":"Yang, Z., Wei, Y., Yang, Y.: Associating objects with transformers for video object segmentation. In: Advances in Neural Information Processing Systems, vol. 34, pp. 2491\u20132502 (2021)"},{"key":"12_CR70","unstructured":"Yang, Z., Yang, Y.: Decoupling features in hierarchical propagation for video object segmentation. In: Advances in Neural Information Processing Systems, vol. 35, pp. 36324\u201336336 (2022)"},{"key":"12_CR71","doi-asserted-by":"crossref","unstructured":"Yao, J., Wang, X., Ye, L., Liu, W.: Matte anything: interactive natural image matting with segment anything models. arXiv preprint arXiv:2306.04121 (2023)","DOI":"10.1016\/j.imavis.2024.105067"},{"key":"12_CR72","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1007\/978-3-319-46475-6_5","volume-title":"Computer Vision \u2013 ECCV 2016","author":"L Yu","year":"2016","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L.: Modeling context in referring expressions. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9906, pp. 69\u201385. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46475-6_5"},{"key":"12_CR73","unstructured":"Yu, T., et al.: Inpaint anything: segment anything meets image inpainting. arXiv preprint arXiv:2304.06790 (2023)"},{"key":"12_CR74","unstructured":"Zhang, J., Cui, Y., Wu, G., Wang, L.: Joint modeling of feature, correspondence, and a compressed memory for video object segmentation. arXiv preprint arXiv:2308.13505 (2023)"},{"key":"12_CR75","unstructured":"Zhao, Q., et al.: Competeai: understanding the competition behaviors in large language model-based agents. arXiv preprint arXiv:2310.17512 (2023)"},{"key":"12_CR76","unstructured":"Zhou, J., et al.: Audio-visual segmentation with semantics. arXiv preprint arXiv:2301.13190 (2023)"},{"key":"12_CR77","doi-asserted-by":"crossref","unstructured":"Zhou, J., et al.: Audio-visual segmentation. In: European Conference on Computer Vision (2022)","DOI":"10.1007\/978-3-031-19836-6_22"},{"key":"12_CR78","doi-asserted-by":"crossref","unstructured":"Zhu, C., Chen, F., Ahmed, U., Shen, Z., Savvides, M.: Semantic relation reasoning for shot-stable few-shot object detection. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8778\u20138787 (2021). https:\/\/api.semanticscholar.org\/CorpusID:232093016","DOI":"10.1109\/CVPR46437.2021.00867"},{"key":"12_CR79","doi-asserted-by":"crossref","unstructured":"Zhu, C., Chen, F., Shen, Z., Savvides, M.: Soft anchor-point object detection. In: European Conference on Computer Vision (2019). https:\/\/api.semanticscholar.org\/CorpusID:208512715","DOI":"10.1007\/978-3-030-58545-7_6"},{"key":"12_CR80","doi-asserted-by":"crossref","unstructured":"Zou, X., et al.: Generalized decoding for pixel, image, and language. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15116\u201315127 (2023)","DOI":"10.1109\/CVPR52729.2023.01451"},{"key":"12_CR81","unstructured":"Zou, X., et al.: Segment everything everywhere all at once. arXiv preprint arXiv:2304.06718 (2023)"},{"key":"12_CR82","unstructured":"Zou, X., et al.: Segment everything everywhere all at once. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72673-6_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T00:03:32Z","timestamp":1732925012000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72673-6_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,22]]},"ISBN":["9783031726729","9783031726736"],"references-count":82,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72673-6_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,10,22]]},"assertion":[{"value":"22 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}