{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T13:46:23Z","timestamp":1782827183032,"version":"3.54.5"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133396","type":"journal-article","created":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:26:14Z","timestamp":1782318374000},"page":"133396","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["Text answer guided RGB-D saliency detection"],"prefix":"10.1016","volume":"331","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1811-7274","authenticated-orcid":false,"given":"Yuxiang","family":"Fu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingkuan","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"4","key":"10.1016\/j.eswa.2026.133396_bib0001","doi-asserted-by":"crossref","first-page":"1787","DOI":"10.1109\/TCSVT.2022.3215979","article-title":"Modality-induced transfer-fusion network for RGB-D and RGB-T salient object detection","volume":"33","author":"Chen","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.133396_bib0002","series-title":"STENet: Superpixel token enhancing network for RGB-D salient object detection","author":"Chen","year":"2026"},{"issue":"3","key":"10.1016\/j.eswa.2026.133396_bib0003","doi-asserted-by":"crossref","first-page":"4309","DOI":"10.1109\/TNNLS.2022.3202241","article-title":"3-D convolutional neural networks for RGB-D salient object detection and beyond","volume":"35","author":"Chen","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"5","key":"10.1016\/j.eswa.2026.133396_bib0004","doi-asserted-by":"crossref","first-page":"2075","DOI":"10.1109\/TNNLS.2020.2996406","article-title":"Rethinking RGB-D salient object detection: Models, datasets, and large-scale benchmarks","volume":"32","author":"Fan","year":"2021","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"2","key":"10.1016\/j.eswa.2026.133396_bib0005","article-title":"M2RNet: Multi-modal and multi-scale refined network for RGB-D salient object detection","volume":"135","author":"Fang","year":"2023","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133396_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108666","article-title":"Encoder deep interleaved network with multi-scale aggregation for RGB-D salient object detection","volume":"128","author":"Feng","year":"2022","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133396_bib0007","doi-asserted-by":"crossref","DOI":"10.1016\/j.envsoft.2025.106629","article-title":"A novel salient object detection network for burned area segmentation in high-resolution remote sensing images","volume":"193","author":"Fu","year":"2025","journal-title":"Environmental Modelling & Software"},{"key":"10.1016\/j.eswa.2026.133396_bib0008","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.129624","article-title":"Incorporating estimated depth maps and multi-modal pretraining to improve salient object detection in optical remote sensing images","volume":"298","author":"Fu","year":"2026","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133396_bib0009","first-page":"1","article-title":"Rethinking masked autoencoder for salient object detection in optical remote sensing images from a pseudo image pretraining perspective","volume":"63","author":"Fu","year":"2025","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.eswa.2026.133396_bib0010","doi-asserted-by":"crossref","first-page":"222","DOI":"10.1016\/j.isprsjprs.2025.03.025","article-title":"Saliency supervised masked autoencoder pretrained salient location mining network for remote sensing image salient object detection","volume":"224","author":"Fu","year":"2025","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"issue":"4","key":"10.1016\/j.eswa.2026.133396_bib0011","doi-asserted-by":"crossref","first-page":"3104","DOI":"10.1109\/TCSVT.2024.3502244","article-title":"Highly efficient RGB-D salient object detection with adaptive fusion and attention regulation","volume":"35","author":"Gao","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.133396_bib0012","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.110190","article-title":"TSVT: Token sparsification vision transformer for robust RGB-D salient object detection","volume":"148","author":"Gao","year":"2024","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133396_bib0013","doi-asserted-by":"crossref","first-page":"7622","DOI":"10.1109\/TMM.2024.3369922","article-title":"Unitr: A unified transformer-based framework for co-object and multi-modal saliency detection","volume":"26","author":"Guo","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133396_bib0014","doi-asserted-by":"crossref","first-page":"2321","DOI":"10.1109\/TIP.2022.3154931","article-title":"Dmra: Depth-induced multi-scale recurrent attention network for RGB-D saliency detection","volume":"31","author":"Ji","year":"2022","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133396_bib0015","series-title":"Proc. the IEEE international conference on image processing","first-page":"1115","article-title":"Depth saliency based on anisotropic center-surround difference","author":"Ju","year":"2014"},{"key":"10.1016\/j.eswa.2026.133396_bib0016","series-title":"Proc. 2024 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"9579","article-title":"LISA: Reasoning segmentation via large language model","author":"Lai","year":"2024"},{"key":"10.1016\/j.eswa.2026.133396_bib0017","series-title":"Proc. ICML","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"issue":"1","key":"10.1016\/j.eswa.2026.133396_bib0018","doi-asserted-by":"crossref","first-page":"479","DOI":"10.1109\/TPAMI.2023.3324807","article-title":"Robust perception and precise segmentation for scribble-supervised RGB-D saliency detection","volume":"46","author":"Li","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133396_bib0019","series-title":"Proc. 2025 IEEE\/CVF international conference on computer vision (ICCV)","first-page":"24056","article-title":"LIRA: Inferring segmentation in large multi-modal models with local interleaved region assistance","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.133396_bib0020","series-title":"Proc. 2024 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"12687","article-title":"Language models as black-box optimizers for vision-language models","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133396_bib0021","series-title":"Advances in neural information processing systems (NeurIPS)","article-title":"UniPixel: Unified object referring and segmentation for pixel-level visual reasoning","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133396_bib0022","series-title":"Proc. CVPR","first-page":"11966","article-title":"A ConVnet for the 2020s","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133396_bib0023","doi-asserted-by":"crossref","DOI":"10.1109\/TIM.2024.3370783","article-title":"Hfmdnet: Hierarchical fusion and multilevel decoder network for RGB-D salient object detection","volume":"73","author":"Luo","year":"2024","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.133396_bib0024","series-title":"Proc. IEEE\/CVF conference on computer vision and pattern recognition","first-page":"454","article-title":"Leveraging stereopsis for saliency analysis","author":"Niu","year":"2012"},{"key":"10.1016\/j.eswa.2026.133396_bib0025","doi-asserted-by":"crossref","first-page":"892","DOI":"10.1109\/TIP.2023.3234702","article-title":"Caver: Cross-modal view-mixed transformer for bi-modal salient object detection","volume":"32","author":"Pang","year":"2023","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133396_bib0026","series-title":"Proc. European conference on computer vision","first-page":"92","article-title":"Rgbd salient object detection: A benchmark and algorithms","author":"Peng","year":"2014"},{"key":"10.1016\/j.eswa.2026.133396_bib0027","series-title":"Proc. ICML","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.133396_bib0028","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107244","article-title":"Bio-inspired two-stage network for efficient RGB-D salient object detection","volume":"185","author":"Ren","year":"2025","journal-title":"Neural Networks"},{"key":"10.1016\/j.eswa.2026.133396_bib0029","series-title":"Proc. 2022 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"8312","article-title":"NLX-GPT: A model for natural language explanations in vision and vision-language tasks","author":"Sammani","year":"2022"},{"key":"10.1016\/j.eswa.2026.133396_bib0030","series-title":"Proc. 2025 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"19769","article-title":"FastVLM: Efficient vision encoding for vision language models","author":"Vasu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133396_bib0031","series-title":"RSONet: Region-guided selective optimization network for rgb-t salient object detection","author":"Wan","year":"2026"},{"issue":"5044416","key":"10.1016\/j.eswa.2026.133396_bib0032","article-title":"Cognition-inspired dynamic feature integration network for RGB-D and RGB-T salient object detection","volume":"74","author":"Wang","year":"2025","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.133396_bib0033","series-title":"Proc. CVPR","first-page":"2830","article-title":"Multilateral semantic relations modeling for image text retrieval","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133396_bib0034","doi-asserted-by":"crossref","first-page":"10342","DOI":"10.1109\/TMM.2024.3407664","article-title":"Estimating the semantics via sector embedding for image-text retrieval","volume":"26","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133396_bib0035","doi-asserted-by":"crossref","first-page":"2226","DOI":"10.1109\/TIP.2024.3374111","article-title":"Semantics disentangling for cross-modal retrieval","volume":"33","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133396_bib0036","series-title":"Distribution-to-points matching for image text retrieval","author":"Wang","year":"2026"},{"key":"10.1016\/j.eswa.2026.133396_bib0037","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2025.105835","article-title":"Multi-modal cooperative fusion network for dual-stream RGB-D salient object detection","volume":"166","author":"Wu","year":"2026","journal-title":"Image and Vision Computing"},{"key":"10.1016\/j.eswa.2026.133396_bib0038","doi-asserted-by":"crossref","first-page":"2160","DOI":"10.1109\/TIP.2023.3263111","article-title":"HiDANet: RGB-D salient object detection via hierarchical depth awareness","volume":"32","author":"Wu","year":"2023","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133396_bib0039","series-title":"Proc. 2024 IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"4818","article-title":"Florence-2: Advancing a unified representation for a variety of vision tasks","author":"Xiao","year":"2024"},{"key":"10.1016\/j.eswa.2026.133396_bib0040","series-title":"Proc. NeurIPS","first-page":"1166","article-title":"Depth anything V2","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133396_bib0041","doi-asserted-by":"crossref","first-page":"2477","DOI":"10.1109\/TMM.2024.3521699","article-title":"FasterSal: Robust and real-time single-stream architecture for RGB-D salient object detection","volume":"27","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133396_bib0042","series-title":"Proc. AAAI conf. artif. intell.","first-page":"3463","article-title":"Self-supervised pretraining for rgb-d salient object detection","author":"Zhao","year":"2022"},{"key":"10.1016\/j.eswa.2026.133396_bib0043","series-title":"NESS-Net: An NAMLab edge-guided and scribble-supervised swin-transformer net for RGB-D salient object detection","author":"Zheng","year":"2026"},{"key":"10.1016\/j.eswa.2026.133396_bib0044","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109459","article-title":"RMFDNet: Redundant and missing feature decoupling network for salient object detection","volume":"139","author":"Zhou","year":"2025","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.133396_bib0045","doi-asserted-by":"crossref","first-page":"4132","DOI":"10.1109\/TNNLS.2021.3105484","article-title":"IRFR-Net: Interactive recursive feature-reshaping network for detecting salient objects in RGB-D images","volume":"36","author":"Zhou","year":"2025","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.133396_bib0046","doi-asserted-by":"crossref","first-page":"4341","DOI":"10.1109\/TASE.2024.3410182","article-title":"MSNet: Multiple strategy network with bidirectional fusion for detecting salient objects in RGB-D images","volume":"22","author":"Zhou","year":"2025","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"key":"10.1016\/j.eswa.2026.133396_bib0047","doi-asserted-by":"crossref","first-page":"495","DOI":"10.1109\/TIP.2025.3648880","article-title":"Turbidity-similarity decoupling: Feature-consistent mutual learning for underwater salient object detection","volume":"35","author":"Zhou","year":"2026","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133396_bib0048","doi-asserted-by":"crossref","first-page":"2192","DOI":"10.1109\/TMM.2021.3077767","article-title":"CCAFNet: Crossflow and cross-scale adaptive fusion network for detecting salient objects in RGB-D images","volume":"24","author":"Zhou","year":"2022","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133396_bib0049","doi-asserted-by":"crossref","first-page":"676","DOI":"10.1109\/TMM.2021.3129730","article-title":"S3Net: Self-supervised self-ensembling network for semi-supervised RGB-D salient object detection","volume":"25","author":"Zhu","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133396_bib0050","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112910","article-title":"Scenario potentiality-constrain network for RGB-D salient object detection","volume":"vol. 310","author":"Zong","year":"2025","journal-title":"Knowledge-Based Systems"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426023055?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426023055?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T13:07:08Z","timestamp":1782824828000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426023055"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":50,"alternative-id":["S0957417426023055"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133396","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Text answer guided RGB-D saliency detection","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133396","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133396"}}