{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T21:02:17Z","timestamp":1784408537505,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":20,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234097","type":"print"},{"value":"9789819234103","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3410-3_41","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T20:09:56Z","timestamp":1784405396000},"page":"485-496","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["PSGF: Progressive Semantic-Guided Fusion for Ambiguity-Aware 3D Visual Grounding"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-4175-3609","authenticated-orcid":false,"given":"Zhuangzhi","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1875-7712","authenticated-orcid":false,"given":"Jinyuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3145-5768","authenticated-orcid":false,"given":"Yunjing","family":"Yi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8719-3938","authenticated-orcid":false,"given":"Yang","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3554-5549","authenticated-orcid":false,"given":"Yundong","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1707-5685","authenticated-orcid":false,"given":"Jinhe","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"41_CR1","doi-asserted-by":"crossref","unstructured":"Liu, D., Liu, Y., Huang, W., et al.: A survey on text-guided 3D visual grounding: elements, recent advances, and future directions. IEEE Trans. Neural Networks Learn. Syst. (2025)","DOI":"10.1109\/TNNLS.2025.3584895"},{"key":"41_CR2","first-page":"202","volume-title":"European Conference on Computer Vision","author":"DZ Chen","year":"2020","unstructured":"Chen, D.Z., Chang, A.X., Nie\u00dfner, M.: ScanRefer: 3D object localization in RGB-D scans using natural language. In: European Conference on Computer Vision, pp. 202\u2013221. Springer, Cham (2020)"},{"key":"41_CR3","first-page":"1046","volume-title":"Conference on Robot Learning","author":"J Roh","year":"2022","unstructured":"Roh, J., Desingh, K., Farhadi, A., et al.: LanguageRefer: spatial-language model for 3D visual grounding. In: Conference on Robot Learning, pp. 1046\u20131056. PMLR (2022)"},{"key":"41_CR4","first-page":"422","volume-title":"European Conference on Computer Vision","author":"P Achlioptas","year":"2020","unstructured":"Achlioptas, P., Abdelreheem, A., Xia, F., et al.: ReferIt3D: neural listeners for fine-grained 3D object identification in real-world scenes. In: European Conference on Computer Vision, pp. 422\u2013440. Springer, Cham (2020)"},{"key":"41_CR5","first-page":"1791","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Z Yuan","year":"2021","unstructured":"Yuan, Z., Yan, X., Liao, Y., et al.: InstanceRefer: Cooperative holistic understanding for visual grounding on point clouds through instance multi-level contextual referring. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1791\u20131800 (2021)"},{"key":"41_CR6","first-page":"417","volume-title":"European Conference on Computer Vision","author":"A Jain","year":"2022","unstructured":"Jain, A., Gkanatsios, N., Mediratta, I., et al.: Bottom up top down detection transformers for language grounding in images and point clouds. In: European Conference on Computer Vision, pp. 417\u2013433. Springer, Cham (2022)"},{"key":"41_CR7","first-page":"19231","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Wu","year":"2023","unstructured":"Wu, Y., Cheng, X., Zhang, R., et al.: EDA: Explicit text-decoupling and dense alignment for 3D visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19231\u201319242 (2023)"},{"key":"41_CR8","first-page":"3666","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"W Guo","year":"2025","unstructured":"Guo, W., Xu, X., Wang, Z., et al.: Text-guided sparse voxel pruning for efficient 3D visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3666\u20133675 (2025)"},{"issue":"7","key":"41_CR9","first-page":"7296","volume":"38","author":"T Zhang","year":"2024","unstructured":"Zhang, T., He, S., Dai, T., et al.: Vision-language pre-training with object contrastive learning for 3D scene understanding. Proceed. AAAI Conf. Artif. Intell. 38(7), 7296\u20137304 (2024)","journal-title":"Proceed. AAAI Conf. Artif. Intell."},{"key":"41_CR10","unstructured":"Liu, Y., Ott, M., Goyal, N., et al.: RoBERTa: A robustly optimized BERT pretraining approach. arXiv preprint https:\/\/arxiv.org\/abs\/1907.11692 (2019)"},{"key":"41_CR11","first-page":"5828","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"A Dai","year":"2017","unstructured":"Dai, A., Chang, A.X., Savva, M., et al.: ScanNet: Richly-annotated 3D reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5828\u20135839 (2017)"},{"key":"41_CR12","first-page":"2928","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"L Zhao","year":"2021","unstructured":"Zhao, L., Cai, D., Sheng, L., et al.: 3DVG-Transformer: Relation modeling for visual grounding on point clouds. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2928\u20132937 (2021)"},{"key":"41_CR13","first-page":"3722","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"M Feng","year":"2021","unstructured":"Feng, M., Li, Z., Li, Q., et al.: Free-form description guided 3D visual graph network for object grounding in point cloud. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3722\u20133731 (2021)"},{"key":"41_CR14","unstructured":"Qi, C.R., Yi, L., Su, H., et al.: PointNet++: deep hierarchical feature learning on point sets in a metric space. Adv. Neural Inf. Proces. Syst. 30 (2017)"},{"key":"41_CR15","first-page":"3075","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"C Choy","year":"2019","unstructured":"Choy, C., Gwak, J.Y., Savarese, S.: 4D spatio-temporal convnets: Minkowski convolutional neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3075\u20133084 (2019)"},{"key":"41_CR16","unstructured":"Graham, B.: Sparse 3D convolutional neural networks. arXiv preprint https:\/\/arxiv.org\/abs\/1505.02890 (2015)"},{"issue":"2\u20133","key":"41_CR17","doi-asserted-by":"publisher","first-page":"217","DOI":"10.1177\/0278364919897133","volume":"39","author":"M Shridhar","year":"2020","unstructured":"Shridhar, M., Mittal, D., Hsu, D.: INGRESS: interactive visual grounding of referring expressions. Int. J. Robot. Res. 39(2\u20133), 217\u2013232 (2020)","journal-title":"Int. J. Robot. Res."},{"key":"41_CR18","unstructured":"Anderson, P., Chang, A., Chaplot, D.S., et al.: On evaluation of embodied navigation agents. arXiv preprint https:\/\/arxiv.org\/abs\/1807.06757 (2018)"},{"issue":"4","key":"41_CR19","doi-asserted-by":"publisher","first-page":"525","DOI":"10.1177\/0018720816644364","volume":"58","author":"TB Sheridan","year":"2016","unstructured":"Sheridan, T.B.: Human\u2013robot interaction: status and challenges. Hum. Factors. 58(4), 525\u2013532 (2016)","journal-title":"Hum. Factors"},{"key":"41_CR20","first-page":"9701","volume-title":"2020 IEEE International Conference on Robotics and Automation","author":"C Gan","year":"2020","unstructured":"Gan, C., Zhang, Y., Wu, J., et al.: Look, listen, and act: towards audio-visual embodied navigation. In: 2020 IEEE International Conference on Robotics and Automation, pp. 9701\u20139707. IEEE (2020)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3410-3_41","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T20:09:59Z","timestamp":1784405399000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3410-3_41"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819234097","9789819234103"],"references-count":20,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3410-3_41","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}