{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T06:45:33Z","timestamp":1783147533301,"version":"3.54.6"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["42501420"],"award-info":[{"award-number":["42501420"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2025M770251"],"award-info":[{"award-number":["2025M770251"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["GZC20252339"],"award-info":[{"award-number":["GZC20252339"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.eswa.2026.131570","type":"journal-article","created":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T17:20:05Z","timestamp":1770657605000},"page":"131570","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":2,"special_numbering":"C","title":["A novel implicit cross-attention framework for RGB-T object detection"],"prefix":"10.1016","volume":"314","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-3123-8547","authenticated-orcid":false,"given":"Zinan","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1290-6016","authenticated-orcid":false,"given":"Chunyu","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6672-367X","authenticated-orcid":false,"given":"Yachao","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6979-3032","authenticated-orcid":false,"given":"Pei","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.131570_bib0001","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"403","article-title":"Multimodal object detection by channel switching and spatial attention","author":"Cao","year":"2023"},{"key":"10.1016\/j.eswa.2026.131570_bib0002","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8628","article-title":"Learning continuous image representation with local implicit image function","author":"Chen","year":"2021"},{"key":"10.1016\/j.eswa.2026.131570_bib0003","series-title":"European conference on computer vision","first-page":"170","article-title":"Transformers as meta-learners for implicit neural representations","author":"Chen","year":"2022"},{"key":"10.1016\/j.eswa.2026.131570_bib0004","doi-asserted-by":"crossref","DOI":"10.1016\/j.asoc.2024.111392","article-title":"Dnnam: Image inpainting algorithm via deep neural networks and attention mechanism","volume":"154","author":"Chen","year":"2024","journal-title":"Applied Soft Computing"},{"key":"10.1016\/j.eswa.2026.131570_bib0005","series-title":"European conference on computer vision","first-page":"139","article-title":"Multimodal object detection via probabilistic ensembling","author":"Chen","year":"2022"},{"key":"10.1016\/j.eswa.2026.131570_bib0006","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5939","article-title":"Learning implicit fields for generative shape modeling","author":"Chen","year":"2019"},{"key":"10.1016\/j.eswa.2026.131570_bib0007","doi-asserted-by":"crossref","first-page":"7392","DOI":"10.1109\/TMM.2025.3599020","article-title":"Fusion-mamba for cross-modality object detection","volume":"27","author":"Dong","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131570_bib0008","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"7132","article-title":"Squeeze-and-excitation networks","author":"Hu","year":"2018"},{"key":"10.1016\/j.eswa.2026.131570_bib0009","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10468","article-title":"Attention convolutional binary neural tree for fine-grained visual categorization","author":"Ji","year":"2020"},{"key":"10.1016\/j.eswa.2026.131570_bib0010","doi-asserted-by":"crossref","first-page":"144","DOI":"10.1016\/j.patrec.2024.02.012","article-title":"Crossformer: Cross-guided attention for multi-modal object detection","volume":"179","author":"Lee","year":"2024","journal-title":"Pattern Recognition Letters"},{"key":"10.1016\/j.eswa.2026.131570_bib0011","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2025.105468","article-title":"Joint transformer and mamba fusion for multispectral object detection","volume":"156","author":"Li","year":"2025","journal-title":"Image and Vision Computing"},{"key":"10.1016\/j.eswa.2026.131570_bib0012","article-title":"Crossmodalnet: A dual-modal object detection network based on cross-modal fusion and channel interaction","volume":"298","author":"Li","year":"2026","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.131570_bib0013","article-title":"Crossmodalnet: A dual-modal object detection network based on cross-modal fusion and channel interaction","volume":"298","author":"Li","year":"2026","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.131570_bib0014","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"4471-4479","article-title":"Learning a graph neural network with cross modality interaction for image fusion","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.131570_bib0015","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"4797","article-title":"Fd2-net: Frequency-driven feature decomposition network for infrared-visible object detection","volume":"vol. 39","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.131570_bib0016","series-title":"Advances in neural information processing systems","first-page":"63441","article-title":"Fourier-enhanced implicit neural fusion network for multispectral and hyperspectral image fusion","volume":"vol. 37","author":"Liang","year":"2024"},{"key":"10.1016\/j.eswa.2026.131570_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103414","article-title":"Como: Cross-mamba interaction and offset-guided fusion for multimodal object detection","volume":"125","author":"Liu","year":"2026","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.131570_bib0018","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5802","article-title":"Target-aware dual adversarial learning and a multi-scenario multi-modality benchmark to fuse infrared and visible for object detection","author":"Liu","year":"2022"},{"issue":"1","key":"10.1016\/j.eswa.2026.131570_bib0019","doi-asserted-by":"crossref","first-page":"315","DOI":"10.1109\/TCSVT.2021.3060162","article-title":"Deep cross-modal representation learning and distillation for illumination-invariant pedestrian detection","volume":"32","author":"Liu","year":"2021","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.131570_bib0020","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108786","article-title":"Cross-modality attentive feature fusion for object detection in multispectral remote sensing imagery","volume":"130","author":"Qingyun","year":"2022","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.131570_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109913","article-title":"Icafusion: Iterative cross-attention guided feature fusion for multispectral object detection","volume":"145","author":"Shen","year":"2024","journal-title":"Pattern Recognition"},{"issue":"1","key":"10.1016\/j.eswa.2026.131570_bib0022","doi-asserted-by":"crossref","first-page":"950","DOI":"10.1109\/TITS.2024.3495028","article-title":"Specificity-guided cross-modal feature reconstruction for RGB-infrared object detection","volume":"26","author":"Sun","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.eswa.2026.131570_bib0023","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"4003-4011","article-title":"Detfusion: A detection-driven infrared and visible image fusion network","author":"Sun","year":"2022"},{"key":"10.1016\/j.eswa.2026.131570_bib0024","series-title":"Multimodal image exploitation and learning 2023","first-page":"165","article-title":"Rgb and ir imagery fusion for autonomous driving","volume":"vol. 12526","author":"Swamy","year":"2023"},{"key":"10.1016\/j.eswa.2026.131570_bib0025","series-title":"Esann","first-page":"509","article-title":"Multispectral pedestrian detection using deep fusion convolutional neural networks","volume":"vol. 587","author":"Wagner","year":"2016"},{"key":"10.1016\/j.eswa.2026.131570_bib0026","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11534","article-title":"Eca-net: Efficient channel attention for deep convolutional neural networks","author":"Wang","year":"2020"},{"key":"10.1016\/j.eswa.2026.131570_bib0027","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111041","article-title":"Mmae: A universal image fusion method via mask attention mechanism","volume":"158","author":"Wang","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.131570_bib0028","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.127110","article-title":"A visible-infrared clothes-changing dataset for person re-identification in natural scene","volume":"569","author":"Wei","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.131570_bib0029","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision (ICCV)","first-page":"9342","article-title":"Vector-decomposed disentanglement for domain-invariant object detection","author":"Wu","year":"2021"},{"key":"10.1016\/j.eswa.2026.131570_bib0030","doi-asserted-by":"crossref","first-page":"8933","DOI":"10.1109\/JSTARS.2023.3315544","article-title":"Cross-modal local calibration and global context modeling network for RGB-infrared remote-sensing object detection","volume":"16","author":"Xie","year":"2023","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"10.1016\/j.eswa.2026.131570_bib0031","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.102939","article-title":"Efficient multispectral object detection with attentive feature aggregation leveraging zero-shot implicit illumination guidance","volume":"118","author":"Xiong","year":"2025","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.131570_bib0032","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19679","article-title":"Rfnet: Unsupervised network for mutually reinforcing multi-modal image registration and fusion","author":"Xu","year":"2022"},{"key":"10.1016\/j.eswa.2026.131570_bib0033","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.129256","article-title":"Clinical CT image super-resolution using pixel-wise hybrid high-dimension mapping-based implicit neural representation","volume":"296","author":"Xue","year":"2026","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.131570_bib0034","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.129226","article-title":"A multi-scale feature extraction and attention aggregation network for underwater image enhancement","volume":"297","author":"Yan","year":"2026","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.131570_bib0035","first-page":"13304","article-title":"Implicit transformer network for screen content image continuous super-resolution","volume":"34","author":"Yang","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.131570_bib0036","series-title":"European conference on computer vision","first-page":"509","article-title":"Translation, scale and rotation: Cross-modal alignment meets RGB-infrared vehicle detection","author":"Yuan","year":"2022"},{"key":"10.1016\/j.eswa.2026.131570_bib0037","series-title":"2020\u202fIEEE International conference on image processing (ICIP)","first-page":"276","article-title":"Multispectral fusion for object detection with cyclic fuse-and-refine blocks","author":"Zhang","year":"2020"},{"key":"10.1016\/j.eswa.2026.131570_bib0038","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"72","article-title":"Guided attentive feature fusion for multispectral pedestrian detection","author":"Zhang","year":"2021"},{"issue":"10","key":"10.1016\/j.eswa.2026.131570_bib0039","doi-asserted-by":"crossref","first-page":"2761","DOI":"10.1007\/s11263-021-01501-8","article-title":"Sdnet: A versatile squeeze-and-decomposition network for real-time image fusion","volume":"129","author":"Zhang","year":"2021","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.131570_bib0040","doi-asserted-by":"crossref","first-page":"20","DOI":"10.1016\/j.inffus.2018.09.015","article-title":"Cross-modality interactive attention network for multispectral pedestrian detection","volume":"50","author":"Zhang","year":"2019","journal-title":"Information Fusion"},{"key":"10.1016\/j.eswa.2026.131570_bib0041","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"10157","article-title":"Enhancing implicit neural representations via symmetric power transformation","volume":"vol. 39","author":"Zhang","year":"2025"},{"issue":"7","key":"10.1016\/j.eswa.2026.131570_bib0042","doi-asserted-by":"crossref","first-page":"13276","DOI":"10.1109\/TNNLS.2024.3443455","article-title":"Tfdet: Target-aware fusion for rgb-t pedestrian detection","volume":"36","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.131570_bib0043","unstructured":"Zhao, T., Yuan, M., & Wei, X. (2024). Removal and selection: Improving RGB-infrared object detection via coarse-to-fine fusion. arXiv preprint arXiv: 2401.10731."},{"key":"10.1016\/j.eswa.2026.131570_bib0044","series-title":"European conference on computer vision","first-page":"787","article-title":"Improving multispectral pedestrian detection by addressing modality imbalance problems","author":"Zhou","year":"2020"},{"key":"10.1016\/j.eswa.2026.131570_bib0045","doi-asserted-by":"crossref","first-page":"7444","DOI":"10.1109\/TMM.2025.3599097","article-title":"Cofnet: Contrastive object-aware fusion using box-level masks for multispectral object detection","volume":"27","author":"Zhou","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131570_bib0046","series-title":"2023\u202fIEEE\/RSJ International conference on intelligent robots and systems (IROS)","first-page":"10918","article-title":"Inf: Implicit neural fusion for lidar and camera","author":"Zhou","year":"2023"},{"issue":"5","key":"10.1016\/j.eswa.2026.131570_bib0047","doi-asserted-by":"crossref","first-page":"4794","DOI":"10.1109\/TITS.2023.3242651","article-title":"Embedded control gate fusion and attention residual learning for RGB-thermal urban scene parsing","volume":"24","author":"Zhou","year":"2023","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.eswa.2026.131570_bib0048","unstructured":"Zhu, L., Liao, B., Zhang, Q., Wang, X., Liu, W., & Wang, X. (2024). Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint arXiv: 2401.09417."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426004835?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426004835?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T17:50:47Z","timestamp":1778781047000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426004835"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":48,"alternative-id":["S0957417426004835"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131570","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A novel implicit cross-attention framework for RGB-T object detection","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131570","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"131570"}}