{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T01:16:50Z","timestamp":1783127810646,"version":"3.54.6"},"reference-count":58,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"publisher","award":["202506560029"],"award-info":[{"award-number":["202506560029"]}],"id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100008642","name":"Center for Health Design","doi-asserted-by":"publisher","award":["300102244202"],"award-info":[{"award-number":["300102244202"]}],"id":[{"id":"10.13039\/100008642","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2041"],"award-info":[{"award-number":["U21B2041"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFB4301800"],"award-info":[{"award-number":["2023YFB4301800"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132657","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:52:53Z","timestamp":1777571573000},"page":"132657","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["From perception to cognition: Unifying multi-object 3D visual grounding and dense captioning in monocular images"],"prefix":"10.1016","volume":"325","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-0699-3992","authenticated-orcid":false,"given":"Keyu","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6462-6087","authenticated-orcid":false,"given":"Yongle","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1414-6181","authenticated-orcid":false,"given":"Hongkai","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4043-8448","authenticated-orcid":false,"given":"Shijie","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5550-6354","authenticated-orcid":false,"given":"Xiangyu","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0384-3743","authenticated-orcid":false,"given":"Mingtao","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9212-5224","authenticated-orcid":false,"given":"Tiantian","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4444-1820","authenticated-orcid":false,"given":"Shan","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5463-1340","authenticated-orcid":false,"given":"Huansheng","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132657_bib0001","series-title":"European conference on computer vision","first-page":"422","article-title":"Referit3D: Neural listeners for fine-grained 3D object identification in real-world scenes","author":"Achlioptas","year":"2020"},{"key":"10.1016\/j.eswa.2026.132657_bib0002","unstructured":"Akinwande, V., Norouzzadeh, M. S., Willmott, D., Bair, A., Ganesh, M. R., & Kolter, J. Z. (2024). HyperCLIP: Adapting vision-language models with hypernetworks. arXiv preprint arXiv: 2412.16777."},{"key":"10.1016\/j.eswa.2026.132657_bib0003","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4453","article-title":"Hyperbolic image segmentation","author":"Atigh","year":"2022"},{"key":"10.1016\/j.eswa.2026.132657_bib0004","series-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","first-page":"65","article-title":"METEOR: An automatic metric for MT evaluation with improved correlation with human judgments","author":"Banerjee","year":"2005"},{"key":"10.1016\/j.eswa.2026.132657_bib0005","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16464","article-title":"3DJCG: A unified framework for joint dense captioning and visual grounding on 3D point clouds","author":"Cai","year":"2022"},{"issue":"3","key":"10.1016\/j.eswa.2026.132657_bib0006","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3641289","article-title":"A survey on evaluation of large language models","volume":"15","author":"Chang","year":"2024","journal-title":"ACM Transactions on Intelligent Systems and Technology"},{"key":"10.1016\/j.eswa.2026.132657_bib0007","series-title":"European conference on computer vision","first-page":"202","article-title":"ScanRefer: 3D object localization in RGB-D scans using natural language","author":"Chen","year":"2020"},{"key":"10.1016\/j.eswa.2026.132657_bib0008","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11124","article-title":"End-to-end 3D dense captioning with Vote2Cap-DETR","author":"Chen","year":"2023"},{"key":"10.1016\/j.eswa.2026.132657_bib0009","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3193","article-title":"Scan2Cap: Context-aware dense captioning in RGB-D scans","author":"Chen","year":"2021"},{"key":"10.1016\/j.eswa.2026.132657_bib0010","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"18109","article-title":"Unit3D: A unified transformer for 3D dense captioning and visual grounding","author":"Chen","year":"2023"},{"issue":"4","key":"10.1016\/j.eswa.2026.132657_bib0011","first-page":"599","article-title":"The euclidean space","volume":"2","author":"Darmochwa\u0142","year":"1991","journal-title":"Formalized Mathematics"},{"issue":"2","key":"10.1016\/j.eswa.2026.132657_bib0012","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1109\/34.982903","article-title":"Vision for mobile robot navigation: A survey","volume":"24","author":"DeSouza","year":"2002","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132657_bib0013","series-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: Human language technologies, Vol. 1 (Long and short papers)","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132657_bib0014","unstructured":"Franco, L., Mandica, P., Kallidromitis, K., Guillory, D., Li, Y.-T., Darrell, T., & Galasso, F. (2023). Hyperbolic active learning for semantic segmentation under domain shift. arXiv preprint arXiv: 2306.11180."},{"key":"10.1016\/j.eswa.2026.132657_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"6840","article-title":"Hyperbolic contrastive learning for visual representations beyond objects","author":"Ge","year":"2023"},{"issue":"11","key":"10.1016\/j.eswa.2026.132657_bib0016","doi-asserted-by":"crossref","first-page":"1231","DOI":"10.1177\/0278364913491297","article-title":"Vision meets robotics: The kitti dataset","volume":"32","author":"Geiger","year":"2013","journal-title":"The International Journal of Robotics Research"},{"key":"10.1016\/j.eswa.2026.132657_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.127273","article-title":"Edge intelligence for intelligent transport systems: Approaches, challenges, and future directions","volume":"280","author":"Ghasemi","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132657_bib0018","unstructured":"Gong, J. (2024). MiniMind-V: Train a 26M-parameter vision-language model from scratch in one hour. https:\/\/github.com\/jingyaogong\/minimind-v. GitHub repository, accessed 2025-07-12."},{"key":"10.1016\/j.eswa.2026.132657_bib0019","doi-asserted-by":"crossref","DOI":"10.1016\/j.compeleceng.2025.110116","article-title":"SmartTrack: Sparse multiple objects association with selective re-identification tracking","volume":"123","author":"Guo","year":"2025","journal-title":"Computers and Electrical Engineering"},{"key":"10.1016\/j.eswa.2026.132657_bib0020","series-title":"2025 International joint conference on neural networks (IJCNN)","first-page":"1","article-title":"MSALNet: Capturing contextual relationships for monocular 3D visual grounding","author":"Guo","year":"2025"},{"key":"10.1016\/j.eswa.2026.132657_bib0021","article-title":"Visual grounding in 2D and 3D: A unified perspective and survey","volume":"126","author":"Guo","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.132657_bib0022","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"3751","article-title":"Beyond human perception: Understanding multi-object world from monocular view","author":"Guo","year":"2025"},{"issue":"1","key":"10.1016\/j.eswa.2026.132657_bib0023","doi-asserted-by":"crossref","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","article-title":"A Survey on Vision Transformer","volume":"45","author":"Han","year":"2022","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132657_bib0024","series-title":"European conference on computer vision","first-page":"367","article-title":"TOD3Cap: Towards 3D dense captioning in outdoor scenes","author":"Jin","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0025","doi-asserted-by":"crossref","first-page":"11703","DOI":"10.52202\/075280-0514","article-title":"MonoUNI: A unified vehicle and infrastructure-side monocular 3D object detection network with sufficient depth clues","volume":"36","author":"Jinrang","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"1","key":"10.1016\/j.eswa.2026.132657_bib0026","doi-asserted-by":"crossref","DOI":"10.1049\/ipr2.13315","article-title":"Bootstrapping vision\u2013language transformer for monocular 3D visual grounding","volume":"19","author":"Lei","year":"2025","journal-title":"IET Image Processing"},{"key":"10.1016\/j.eswa.2026.132657_bib0027","series-title":"Text summarization branches out","first-page":"74","article-title":"ROUGE: A package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.eswa.2026.132657_bib0028","series-title":"European conference on computer vision","first-page":"456","article-title":"WildRefer: 3D object localization in large-scale dynamic scenes with multi-modal visual data and natural language","author":"Lin","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0029","unstructured":"Liu, A., Feng, B., Xue, B., Wang, B., Wu, B., Lu, C., Zhao, C., Deng, C., Zhang, C., Ruan, C. et al. (2024a). DeepSeek-v3 technical report. arXiv preprint arXiv: 2412.19437."},{"key":"10.1016\/j.eswa.2026.132657_bib0030","unstructured":"Liu, D., Liu, Y., Huang, W., & Hu, W. (2024b). A survey on text-guided 3D visual grounding: elements, recent advances, and future directions. arXiv preprint arXiv: 2406.05785."},{"key":"10.1016\/j.eswa.2026.132657_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"6032","article-title":"Refer-it-in-RGBD: A bottom-up approach for 3D visual grounding in RGBD images","author":"Liu","year":"2021"},{"key":"10.1016\/j.eswa.2026.132657_bib0032","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). RoBERTA: A robustly optimized bert pretraining approach. arXiv preprint arXiv: 1907.11692."},{"issue":"9","key":"10.1016\/j.eswa.2026.132657_bib0033","doi-asserted-by":"crossref","first-page":"3484","DOI":"10.1007\/s11263-024-02043-5","article-title":"Hyperbolic deep learning in computer vision: A survey","volume":"132","author":"Mettes","year":"2024","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.132657_bib0034","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3744746","article-title":"A comprehensive overview of large language models","volume":"16","author":"Naveed","year":"2023","journal-title":"ACM Transactions on Intelligent Systems and Technology"},{"key":"10.1016\/j.eswa.2026.132657_bib0035","unstructured":"Pal, A., van Spengler, M., di Melendugno, G. M. D., Flaborea, A., Galasso, F., & Mettes, P. (2024). Compositional entailment learning for hyperbolic vision-language models. arXiv preprint arXiv: 2410.06912."},{"key":"10.1016\/j.eswa.2026.132657_bib0036","series-title":"Proceedings of the 40th annual meeting of the association for computational linguistics","first-page":"311","article-title":"BLEU: A method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"key":"10.1016\/j.eswa.2026.132657_bib0037","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132657_bib0038","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"27263","article-title":"Accept the modality gap: An exploration in the hyperbolic space","author":"Ramasinghe","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0039","series-title":"International conference on data intelligence and cognitive informatics","first-page":"529","article-title":"A review on yolov8 and its advancements","author":"Sohan","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0040","unstructured":"Sturua, S., Mohr, I., Akram, M. K., G\u00fcnther, M., Wang, B., Krimmel, M., Wang, F., Mastrapas, G., Koukounas, A., Wang, N. et al. (2024). Jina-embeddings-v3: Multilingual embeddings with task LoRA. arXiv preprint arXiv: 2409.10173."},{"key":"10.1016\/j.eswa.2026.132657_bib0041","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128841","article-title":"Safe navigation for robotic digestive endoscopy via human intervention-based reinforcement learning","volume":"294","author":"Tan","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132657_bib0042","first-page":"4566","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132657_bib0043","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"4566","article-title":"CIDEr: Consensus-based image description evaluation","author":"Vedantam","year":"2015"},{"key":"10.1016\/j.eswa.2026.132657_bib0044","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"1644","article-title":"Machine unlearning in hyperbolic vs. euclidean multimodal contrastive learning: Adapting alignment calibration to MERU","author":"Vidal","year":"2025"},{"key":"10.1016\/j.eswa.2026.132657_bib0045","doi-asserted-by":"crossref","unstructured":"Wang, H., Zhang, C., Yu, J., & Cai, W. (2022). Spatiality-guided transformer for 3D dense captioning on point clouds. arXiv preprint arXiv: 2204.10688.","DOI":"10.24963\/ijcai.2022\/194"},{"key":"10.1016\/j.eswa.2026.132657_bib0046","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13917","article-title":"G\u2303 3-LQ: Marrying hyperbolic alignment with explicit semantic-geometric modeling for 3D visual grounding","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0047","series-title":"2024 3rd International conference on robotics, artificial intelligence and intelligent control (RAIIC)","first-page":"78","article-title":"Research on autonomous robots navigation based on reinforcement learning","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0048","series-title":"Proceedings of the first workshop on evaluation and comparison of NLP systems","first-page":"79","article-title":"Probabilistic extension of precision, recall, and F1 score for more thorough evaluation of classification models","author":"Yacouby","year":"2020"},{"key":"10.1016\/j.eswa.2026.132657_bib0049","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"21341","article-title":"Rope3D: The roadside perception dataset for autonomous driving and monocular 3D object detection task","author":"Ye","year":"2022"},{"issue":"3","key":"10.1016\/j.eswa.2026.132657_bib0050","doi-asserted-by":"crossref","first-page":"1322","DOI":"10.1109\/TCSVT.2023.3296889","article-title":"A comprehensive survey of 3d dense captioning: Localizing and describing objects in 3D scenes","volume":"34","author":"Yu","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132657_bib0051","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"1791","article-title":"InstanceRefer: Cooperative holistic understanding for visual grounding on point clouds through instance multi-level contextual referring","author":"Yuan","year":"2021"},{"key":"10.1016\/j.eswa.2026.132657_bib0052","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"6988","article-title":"Mono3DVG: 3D visual grounding in monocular images","volume":"Vol. 38","author":"Zhan","year":"2024"},{"key":"10.1016\/j.eswa.2026.132657_bib0053","doi-asserted-by":"crossref","unstructured":"Zhang, H., Yang, C.-A., & Yeh, R. A. (2024). Multi-object 3D grounding with dynamic modules and language-informed spatial attention. arXiv preprint arXiv: 2410.22306.","DOI":"10.52202\/079017-3917"},{"key":"10.1016\/j.eswa.2026.132657_bib0054","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15225","article-title":"Multi3DRefer: Grounding text description to multiple 3D objects","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132657_bib0055","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"2928","article-title":"3DVG-transformer: Relation modeling for visual grounding on point clouds","author":"Zhao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132657_bib0056","doi-asserted-by":"crossref","unstructured":"Zhenyu Chen, D., Wu, Q., Nie\u00dfner, M., & Chang, A. X. (2021). D3Net: A unified speaker-listener architecture for 3D dense captioning and visual grounding. arXiv e-prints, pages arXiv\u20132112.","DOI":"10.1007\/978-3-031-19824-3_29"},{"key":"10.1016\/j.eswa.2026.132657_bib0057","doi-asserted-by":"crossref","DOI":"10.1016\/j.sigpro.2024.109749","article-title":"PV-LAP: Multi-sensor fusion for 3D scene understanding in intelligent transportation systems","volume":"227","author":"Zhu","year":"2025","journal-title":"Signal Processing"},{"key":"10.1016\/j.eswa.2026.132657_bib0058","unstructured":"Zong, M., & Krishnamachari, B. (2022). A survey on GPT-3. arXiv preprint arXiv: 2212.00857."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015708?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015708?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:30:24Z","timestamp":1783125024000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426015708"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":58,"alternative-id":["S0957417426015708"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132657","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"From perception to cognition: Unifying multi-object 3D visual grounding and dense captioning in monocular images","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132657","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132657"}}