{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T07:56:37Z","timestamp":1781164597845,"version":"3.54.1"},"reference-count":27,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition Letters"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.patrec.2026.04.034","type":"journal-article","created":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T15:13:40Z","timestamp":1779203620000},"page":"42-46","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Scene-verified negatives for selective 3D visual grounding"],"prefix":"10.1016","volume":"207","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2300-2007","authenticated-orcid":false,"given":"Songyuan","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-5621-2489","authenticated-orcid":false,"given":"Zhihao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3532-3455","authenticated-orcid":false,"given":"Haolan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zixuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patrec.2026.04.034_bib0001","series-title":"Proc. Eur. Conf. Comput. Vis. (ECCV)","article-title":"ScanRefer: 3D object localization in RGB-D scans using natural language","author":"Chen","year":"2020"},{"key":"10.1016\/j.patrec.2026.04.034_bib0002","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"ScanNet: richly-annotated 3D reconstructions of indoor scenes","author":"Dai","year":"2017"},{"key":"10.1016\/j.patrec.2026.04.034_bib0003","series-title":"Proc. Eur. Conf. Comput. Vis. (ECCV)","article-title":"Bottom-up top-down detection transformers for language grounding in images and point clouds","author":"Jain","year":"2022"},{"key":"10.1016\/j.patrec.2026.04.034_bib0004","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"Multi-view transformer for 3D visual grounding","author":"Huang","year":"2022"},{"key":"10.1016\/j.patrec.2026.04.034_bib0005","series-title":"Proceedings of the 29th ACM International Conference on Multimedia","first-page":"1071","article-title":"TransRefer3D: entity-and-relation aware transformer for fine-grained 3D visual grounding","author":"He","year":"2021"},{"key":"10.1016\/j.patrec.2026.04.034_bib0006","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","article-title":"3DVG-transformer: relation modeling for visual grounding on point clouds","author":"Zhao","year":"2021"},{"key":"10.1016\/j.patrec.2026.04.034_bib0007","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"123","article-title":"3D-SPS: single-stage 3D visual grounding via referred point progressive selection","author":"Luo","year":"2022"},{"key":"10.1016\/j.patrec.2026.04.034_bib0008","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"EDA: explicit text-decoupling and dense alignment for 3D visual grounding","author":"Wu","year":"2023"},{"key":"10.1016\/j.patrec.2026.04.034_bib0009","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","article-title":"3D-VisTA: pre-trained transformer for 3D vision and text alignment","author":"Zhu","year":"2023"},{"key":"10.1016\/j.patrec.2026.04.034_bib0010","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"Text-guided sparse voxel pruning for efficient 3D visual grounding","author":"Guo","year":"2025"},{"key":"10.1016\/j.patrec.2026.04.034_bib0011","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","article-title":"InstanceRefer: cooperative holistic understanding for visual grounding on point clouds through instance multi-level contextual referring","author":"Yuan","year":"2021"},{"key":"10.1016\/j.patrec.2026.04.034_bib0012","series-title":"Proc. Annu. Meet. Assoc. Comput. Linguist. (ACL)","article-title":"ViGiL3D: a benchmark for linguistic and generalization challenges in 3D visual grounding","author":"Wang","year":"2025"},{"key":"10.1016\/j.patrec.2026.04.034_bib0013","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"Winoground: probing vision and language models for visio-linguistic compositionality","author":"Thrush","year":"2022"},{"key":"10.1016\/j.patrec.2026.04.034_bib0014","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"Vision\u2013language models do not understand negation","author":"Alhamoud","year":"2025"},{"key":"10.1016\/j.patrec.2026.04.034_bib0015","series-title":"Proc. Eur. Conf. Comput. Vis. (ECCV)","article-title":"ReferIt3D: neural listeners for fine-grained 3D object identification in real-world scenes","author":"Achlioptas","year":"2020"},{"key":"10.1016\/j.patrec.2026.04.034_bib0016","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","article-title":"Multi3DRefer: grounding text description to multiple 3D objects","author":"Zhang","year":"2023"},{"key":"10.1016\/j.patrec.2026.04.034_bib0017","series-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","article-title":"A baseline for detecting misclassified and out-of-distribution examples in neural networks","author":"Hendrycks","year":"2017"},{"key":"10.1016\/j.patrec.2026.04.034_bib0018","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. Workshops (CVPRW)","first-page":"38","article-title":"Measuring calibration in deep learning","author":"Nixon","year":"2019"},{"key":"10.1016\/j.patrec.2026.04.034_bib0019","first-page":"1823","article-title":"Classification with a reject option using a hinge loss","volume":"9","author":"Bartlett","year":"2008","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.patrec.2026.04.034_bib0020","doi-asserted-by":"crossref","first-page":"10","DOI":"10.1016\/j.patrec.2023.02.023","article-title":"Transformer vision-language tracking via proxy token guided cross-modal fusion","volume":"168","author":"Zhao","year":"2023","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.patrec.2026.04.034_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.patrec.2025.03.012","article-title":"GazeViT: a gaze-guided hybrid attention vision transformer for cross-view matching of street-to-aerial images","author":"Hu","year":"2025","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.patrec.2026.04.034_bib0022","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","article-title":"Focal loss for dense object detection","author":"Lin","year":"2017"},{"key":"10.1016\/j.patrec.2026.04.034_bib0023","series-title":"Proc. Int. Conf. Mach. Learn. (ICML)","article-title":"On calibration of modern neural networks","author":"Guo","year":"2017"},{"key":"10.1016\/j.patrec.2026.04.034_bib0024","article-title":"Selective classification for deep neural networks","volume":"vol. 30","author":"Geifman","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)"},{"key":"10.1016\/j.patrec.2026.04.034_bib0025","series-title":"Proc. Int. Conf. Mach. Learn. (ICML)","article-title":"SelectiveNet: a deep neural network with an integrated reject option","author":"Geifman","year":"2019"},{"issue":"1","key":"10.1016\/j.patrec.2026.04.034_bib0026","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1109\/TIT.1970.1054406","article-title":"On optimum recognition error and reject tradeoff","volume":"16","author":"Chow","year":"1970","journal-title":"IEEE Trans. Inf. Theory"},{"key":"10.1016\/j.patrec.2026.04.034_bib0027","series-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","article-title":"LoRA: low-rank adaptation of large language models","author":"Hu","year":"2022"}],"container-title":["Pattern Recognition Letters"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001625?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001625?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T06:59:31Z","timestamp":1781161171000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167865526001625"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":27,"alternative-id":["S0167865526001625"],"URL":"https:\/\/doi.org\/10.1016\/j.patrec.2026.04.034","relation":{},"ISSN":["0167-8655"],"issn-type":[{"value":"0167-8655","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Scene-verified negatives for selective 3D visual grounding","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition Letters","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patrec.2026.04.034","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}]}}