{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T01:17:28Z","timestamp":1783127848085,"version":"3.54.6"},"reference-count":51,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2133218"],"award-info":[{"award-number":["U2133218"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2024YFC3308300"],"award-info":[{"award-number":["2024YFC3308300"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Fusion"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.inffus.2026.104412","type":"journal-article","created":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T15:52:53Z","timestamp":1776959573000},"page":"104412","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["AVD-MDet : Knowledge-guided heterogeneous audio-visual alignment for small object detection"],"prefix":"10.1016","volume":"134","author":[{"given":"Xuesong","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0975-2316","authenticated-orcid":false,"given":"Baolin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.inffus.2026.104412_bib0001","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"13668","article-title":"QueryDet: cascaded sparse query for accelerating high-resolution small object detection","author":"Yang","year":"2022"},{"key":"10.1016\/j.inffus.2026.104412_bib0002","doi-asserted-by":"crossref","first-page":"2962","DOI":"10.1109\/TIP.2022.3162099","article-title":"CrabNet: fully task-specific feature learning for one-stage object detection","volume":"31","author":"Wang","year":"2022","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.inffus.2026.104412_bib0003","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.inffus.2026.104412_bib0004","doi-asserted-by":"crossref","first-page":"29","DOI":"10.1016\/j.neucom.2023.01.055","article-title":"DKTNet: dual-key transformer network for small object detection","volume":"525","author":"Xu","year":"2023","journal-title":"Neurocomputing"},{"key":"10.1016\/j.inffus.2026.104412_bib0005","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.inffus.2026.104412_bib0006","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","article-title":"The pascal visual object classes (VOC) challenge","volume":"88","author":"Everingham","year":"2010","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.inffus.2026.104412_bib0007","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10781","article-title":"EfficientDet: scalable and efficient object detection","author":"Tan","year":"2020"},{"key":"10.1016\/j.inffus.2026.104412_bib0008","series-title":"2023 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","first-page":"2184","article-title":"AFPN: asymptotic feature pyramid network for object detection","author":"Yang","year":"2023"},{"key":"10.1016\/j.inffus.2026.104412_bib0009","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2022.118665","article-title":"Tiny object detection with context enhancement and feature purification","volume":"211","author":"Xiao","year":"2023","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.inffus.2026.104412_bib0010","doi-asserted-by":"crossref","first-page":"6917","DOI":"10.1109\/TIP.2021.3099733","article-title":"HCE: hierarchical context embedding for region-based object detection","volume":"30","author":"Chen","year":"2021","journal-title":"IEEE Trans. Image Process."},{"issue":"9","key":"10.1016\/j.inffus.2026.104412_bib0011","doi-asserted-by":"crossref","first-page":"1432","DOI":"10.3390\/rs12091432","article-title":"Small-object detection in remote sensing images with end-to-end edge-enhanced GAN and object detector network","volume":"12","author":"Rabbi","year":"2020","journal-title":"Remote Sens."},{"key":"10.1016\/j.inffus.2026.104412_bib0012","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3774","article-title":"Dual super-resolution learning for semantic segmentation","author":"Wang","year":"2020"},{"key":"10.1016\/j.inffus.2026.104412_bib0013","series-title":"European Conference on Computer Vision","first-page":"139","article-title":"Multimodal object detection via probabilistic ensembling","author":"Chen","year":"2022"},{"key":"10.1016\/j.inffus.2026.104412_bib0014","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"4797","article-title":"FD2-Net: frequency-driven feature decomposition network for infrared-visible object detection","volume":"39","author":"Li","year":"2025"},{"key":"10.1016\/j.inffus.2026.104412_bib0015","unstructured":"F.A. Group, Flir thermal dataset for algorithm training, 2023. https:\/\/www.flir.com\/oem\/adas\/adasdataset-form\/."},{"key":"10.1016\/j.inffus.2026.104412_bib0016","series-title":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"776","article-title":"Audio set: an ontology and human-labeled dataset for audio events","author":"Gemmeke","year":"2017"},{"key":"10.1016\/j.inffus.2026.104412_bib0017","series-title":"ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"721","article-title":"VGGSound: a large-scale audio-visual dataset","author":"Chen","year":"2020"},{"key":"10.1016\/j.inffus.2026.104412_bib0018","first-page":"892","article-title":"SoundNet: learning sound representations from unlabeled video","volume":"29","author":"Aytar","year":"2016","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"12","key":"10.1016\/j.inffus.2026.104412_bib0019","doi-asserted-by":"crossref","first-page":"10028","DOI":"10.1109\/TNNLS.2022.3163771","article-title":"Multimodal sparse transformer network for audio-visual speech recognition","volume":"34","author":"Song","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"10","key":"10.1016\/j.inffus.2026.104412_bib0020","doi-asserted-by":"crossref","first-page":"18587","DOI":"10.1109\/TNNLS.2025.3583509","article-title":"DiffCL: a diffusion-based contrastive learning framework with semantic alignment for multimodal recommendations","volume":"36","author":"Song","year":"2025","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.inffus.2026.104412_bib0021","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"9472-9476","article-title":"Multi-scale conformer fusion network for multi-participant behavior analysis","author":"Song","year":"2023"},{"key":"10.1016\/j.inffus.2026.104412_bib0022","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"2847","article-title":"VisDrone-DET2021: the vision meets drone object detection challenge results","author":"Cao","year":"2021"},{"key":"10.1016\/j.inffus.2026.104412_bib0023","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"1257","article-title":"Scale match for tiny person detection","author":"Yu","year":"2020"},{"key":"10.1016\/j.inffus.2026.104412_bib0024","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3974","article-title":"DOTA: a large-scale dataset for object detection in aerial images","author":"Xia","year":"2018"},{"issue":"11","key":"10.1016\/j.inffus.2026.104412_bib0025","first-page":"13467","article-title":"Towards large-scale small object detection: survey and benchmarks","volume":"45","author":"Cheng","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.inffus.2026.104412_bib0026","series-title":"2015 IEEE International Conference on Image Processing (ICIP)","first-page":"3735","article-title":"Orientation robust object detection in aerial images using deep convolutional neural network","author":"Zhu","year":"2015"},{"key":"10.1016\/j.inffus.2026.104412_bib0027","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"2260","article-title":"SeaDronesSee: a maritime benchmark for detecting humans in open water","author":"Varga","year":"2022"},{"key":"10.1016\/j.inffus.2026.104412_bib0028","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"247","article-title":"Audio-visual event localization in unconstrained videos","author":"Tian","year":"2018"},{"key":"10.1016\/j.inffus.2026.104412_bib0029","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4358","article-title":"Learning to localize sound source in visual scenes","author":"Senocak","year":"2018"},{"key":"10.1016\/j.inffus.2026.104412_bib0030","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6419","article-title":"Reasoning-rCNN: unifying adaptive global reasoning into large-scale object detection","author":"Xu","year":"2019"},{"key":"10.1016\/j.inffus.2026.104412_bib0031","first-page":"1","article-title":"DHANet: dual-stream hierarchical interaction networks for multimodal drone object detection","volume":"63","author":"Wu","year":"2025","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.inffus.2026.104412_bib0032","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"18412","article-title":"Self-prompting analogical reasoning for UAV object detection","volume":"39","author":"Li","year":"2025"},{"key":"10.1016\/j.inffus.2026.104412_bib0033","first-page":"1","article-title":"Airborne small target detection method based on multi-modal and adaptive feature fusion","volume":"62","author":"Xu","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.inffus.2026.104412_bib0034","first-page":"1","article-title":"SuperYOLO: super resolution assisted object detection in multimodal remote sensing imagery","volume":"61","author":"Zhang","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.inffus.2026.104412_bib0035","first-page":"1728","article-title":"DQ-DETR: dual query detection transformer for phrase extraction and grounding","volume":"37","author":"Liu","year":"2023","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104412_bib0036","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.inffus.2026.104412_bib0037","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"2980","article-title":"Focal loss for dense object detection","author":"Lin","year":"2017"},{"key":"10.1016\/j.inffus.2026.104412_bib0038","unstructured":"W. Lv, Y. Zhao, Q. Chang, K. Huang, G. Wang, Y. Liu, RT-DETRv2: improved baseline with bag-of-freebies for real-time detection transformer,(2024). arXiv: 2407.17140."},{"key":"10.1016\/j.inffus.2026.104412_bib0039","first-page":"4643","article-title":"RemDet: rethinking efficient model design for uav object detection","volume":"39","author":"Li","year":"2025","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104412_bib0040","unstructured":"Z. Du, Z. Hu, G. Zhao, Y. Jin, H. Ma, Cross-layer feature pyramid transformer for small object detection in aerial images,(2024). arXiv: 2407.19696."},{"key":"10.1016\/j.inffus.2026.104412_bib0041","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"101","article-title":"Dome-DETR: DETR with density-oriented feature-query manipulation for efficient tiny object detection","author":"Hu","year":"2025"},{"key":"10.1016\/j.inffus.2026.104412_bib0042","unstructured":"Y. Tian, Q. Ye, D. Doermann, YOLO12: attention-centric real-time object detectors, (2025). arXiv: 2502.12524."},{"key":"10.1016\/j.inffus.2026.104412_bib0043","series-title":"2025 International Joint Conference on Neural Networks (IJCNN)","first-page":"1","article-title":"SO-DETR: leveraging dual-domain features and knowledge distillation for small object detection","author":"Zhang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104412_bib0044","series-title":"2025 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"4703","article-title":"MI-DETR: an object detection model with multi-time inquiries mechanism","author":"Nan","year":"2025"},{"issue":"2","key":"10.1016\/j.inffus.2026.104412_bib0045","doi-asserted-by":"crossref","first-page":"1050","DOI":"10.1109\/TPAMI.2020.3013717","article-title":"Prior guided feature enrichment network for few-shot segmentation","volume":"44","author":"Tian","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.inffus.2026.104412_bib0046","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","article-title":"Film: visual reasoning with a general conditioning layer","volume":"32","author":"Perez","year":"2018"},{"key":"10.1016\/j.inffus.2026.104412_bib0047","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"5615","article-title":"Rethink cross-modal fusion in weakly-supervised audio-visual video parsing","author":"Xu","year":"2024"},{"key":"10.1016\/j.inffus.2026.104412_bib0048","series-title":"European Conference on Computer Vision","first-page":"218","article-title":"Localizing visual sounds the easy way","author":"Mo","year":"2022"},{"key":"10.1016\/j.inffus.2026.104412_bib0049","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"26202","article-title":"Learning audio-guided video representation with gated attention for video-text retrieval","author":"Jeong","year":"2025"},{"key":"10.1016\/j.inffus.2026.104412_bib0050","unstructured":"Jocher, G., Chaurasia, A., Qiu, J., 2023. Ultralytics YOLO. https:\/\/github.com\/ultralytics\/ultralytics."},{"key":"10.1016\/j.inffus.2026.104412_bib0051","unstructured":"Jocher, G., Qiu, J., 2026. Ultralytics YOLO26. https:\/\/github.com\/ultralytics\/ultralytics."}],"container-title":["Information Fusion"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526002915?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526002915?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:35:29Z","timestamp":1783125329000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1566253526002915"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":51,"alternative-id":["S1566253526002915"],"URL":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104412","relation":{},"ISSN":["1566-2535"],"issn-type":[{"value":"1566-2535","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"AVD-MDet : Knowledge-guided heterogeneous audio-visual alignment for small object detection","name":"articletitle","label":"Article Title"},{"value":"Information Fusion","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104412","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104412"}}