{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T14:46:38Z","timestamp":1782485198738,"version":"3.54.5"},"reference-count":68,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100010814","name":"Anhui Province Department of Education","doi-asserted-by":"publisher","award":["KJ2019A0022918005"],"award-info":[{"award-number":["KJ2019A0022918005"]}],"id":[{"id":"10.13039\/501100010814","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100019413","name":"Yunnan Provincial Department of Education Science Research Fund Project","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100019413","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003995","name":"Anhui Provincial Natural Science Foundation","doi-asserted-by":"publisher","award":["1908085MF217"],"award-info":[{"award-number":["1908085MF217"]}],"id":[{"id":"10.13039\/501100003995","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.engappai.2026.115425","type":"journal-article","created":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T06:25:43Z","timestamp":1781763943000},"page":"115425","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P2","title":["Hierarchical Spatial\u2013Temporal Feature Integration Network for visual tracking"],"prefix":"10.1016","volume":"181","author":[{"given":"Hua","family":"Bao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongchao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aoshen","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2026.115425_b1","doi-asserted-by":"crossref","unstructured":"Bertinetto, L., Valmadre, J., Golodetz, S., Miksik, O., Torr, P.H., 2016a. Staple: Complementary learners for real-time tracking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1401\u20131409.","DOI":"10.1109\/CVPR.2016.156"},{"key":"10.1016\/j.engappai.2026.115425_b2","series-title":"European Conference on Computer Vision","first-page":"850","article-title":"Fully-convolutional siamese networks for object tracking","author":"Bertinetto","year":"2016"},{"key":"10.1016\/j.engappai.2026.115425_b3","doi-asserted-by":"crossref","unstructured":"Bhat, G., Danelljan, M., Gool, L.V., Timofte, R., 2019. Learning discriminative model prediction for tracking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6182\u20136191.","DOI":"10.1109\/ICCV.2019.00628"},{"key":"10.1016\/j.engappai.2026.115425_b4","doi-asserted-by":"crossref","first-page":"120380","DOI":"10.1109\/ACCESS.2021.3107579","article-title":"A general method for generating discrete orthogonal matrices","volume":"9","author":"Chan","year":"2021","journal-title":"IEEE Access"},{"issue":"1","key":"10.1016\/j.engappai.2026.115425_b5","doi-asserted-by":"crossref","first-page":"011112","DOI":"10.1117\/1.3556727","article-title":"Using admittance spectroscopy to quantify transport properties of P3HT thin films","volume":"1","author":"Chan","year":"2011","journal-title":"J. Photonics Energy"},{"key":"10.1016\/j.engappai.2026.115425_b6","doi-asserted-by":"crossref","unstructured":"Chen, X., Yan, B., Zhu, J., Wang, D., Yang, X., Lu, H., 2021. Transformer tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8126\u20138135.","DOI":"10.1109\/CVPR46437.2021.00803"},{"key":"10.1016\/j.engappai.2026.115425_b7","doi-asserted-by":"crossref","unstructured":"Chen, Z., Zhong, B., Li, G., Zhang, S., Ji, R., 2020. Siamese box adaptive network for visual tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6668\u20136677.","DOI":"10.1109\/CVPR42600.2020.00670"},{"key":"10.1016\/j.engappai.2026.115425_b8","series-title":"Zero emiss","author":"Cheng","year":"2015"},{"key":"10.1016\/j.engappai.2026.115425_b9","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Bhat, G., Khan, F.S., Felsberg, M., 2019. Atom: Accurate tracking by overlap maximization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4660\u20134669.","DOI":"10.1109\/CVPR.2019.00479"},{"key":"10.1016\/j.engappai.2026.115425_b10","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Bhat, G., Shahbaz Khan, F., Felsberg, M., 2017. Eco: Efficient convolution operators for tracking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 6638\u20136646.","DOI":"10.1109\/CVPR.2017.733"},{"key":"10.1016\/j.engappai.2026.115425_b11","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part V 14","first-page":"472","article-title":"Beyond correlation filters: Learning continuous convolution operators for visual tracking","author":"Danelljan","year":"2016"},{"key":"10.1016\/j.engappai.2026.115425_b12","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.engappai.2026.115425_b13","doi-asserted-by":"crossref","unstructured":"Fan, H., Lin, L., Yang, F., Chu, P., Deng, G., Yu, S., Bai, H., Xu, Y., Liao, C., Ling, H., 2019. Lasot: A high-quality benchmark for large-scale single object tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 5374\u20135383.","DOI":"10.1109\/CVPR.2019.00552"},{"key":"10.1016\/j.engappai.2026.115425_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.107976","article-title":"Efficient object tracking algorithm based on lightweight siamese networks","volume":"133","author":"Feng","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.115425_b15","doi-asserted-by":"crossref","unstructured":"Fu, Z., Liu, Q., Fu, Z., Wang, Y., 2021. Stmtrack: Template-free visual tracking with space\u2013time memory networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13774\u201313783.","DOI":"10.1109\/CVPR46437.2021.01356"},{"key":"10.1016\/j.engappai.2026.115425_b16","doi-asserted-by":"crossref","unstructured":"Guo, D., Wang, J., Cui, Y., Wang, Z., Chen, S., 2020. Siamcar: Siamese fully convolutional classification and regression for visual tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6269\u20136277.","DOI":"10.1109\/CVPR42600.2020.00630"},{"key":"10.1016\/j.engappai.2026.115425_b17","series-title":"The Visual Object Tracking Vot2016 Challenge Results","first-page":"777","volume":"vol. 9914","author":"Hadfield","year":"2016"},{"key":"10.1016\/j.engappai.2026.115425_b18","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R., 2022. Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"10.1016\/j.engappai.2026.115425_b19","doi-asserted-by":"crossref","unstructured":"He, A., Luo, C., Tian, X., Zeng, W., 2018. A twofold siamese network for real-time object tracking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4834\u20134843.","DOI":"10.1109\/CVPR.2018.00508"},{"key":"10.1016\/j.engappai.2026.115425_b20","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J., 2016. Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.engappai.2026.115425_b21","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G., 2018. Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 7132\u20137141.","DOI":"10.1109\/CVPR.2018.00745"},{"issue":"12","key":"10.1016\/j.engappai.2026.115425_b22","doi-asserted-by":"crossref","first-page":"5565","DOI":"10.3390\/s23125565","article-title":"Fusion of multi-modal features to enhance dense video caption","volume":"23","author":"Huang","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.engappai.2026.115425_b23","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.107304","article-title":"TATrack: Target-aware transformer for object tracking","volume":"127","author":"Huang","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"issue":"5","key":"10.1016\/j.engappai.2026.115425_b24","doi-asserted-by":"crossref","first-page":"1562","DOI":"10.1109\/TPAMI.2019.2957464","article-title":"Got-10k: A large high-diversity benchmark for generic object tracking in the wild","volume":"43","author":"Huang","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2026.115425_b25","doi-asserted-by":"crossref","unstructured":"Kiani Galoogahi, H., Fagg, A., Huang, C., Ramanan, D., Lucey, S., 2017. Need for speed: A benchmark for higher frame rate object tracking. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 1125\u20131134.","DOI":"10.1109\/ICCV.2017.128"},{"key":"10.1016\/j.engappai.2026.115425_b26","unstructured":"Kristan, M., Leonardis, A., Matas, J., Felsberg, M., Pflugfelder, R., Cehovin Zajc, L., Vojir, T., Bhat, G., Lukezic, A., Eldesokey, A., et al., 2018. The sixth visual object tracking vot2018 challenge results. In: Proceedings of the European Conference on Computer Vision. 0\u20130."},{"key":"10.1016\/j.engappai.2026.115425_b27","unstructured":"Kristan, M., Matas, J., Leonardis, A., Felsberg, M., Pflugfelder, R., Kamarainen, J.-K., Cehovin Zajc, L., Drbohlav, O., Lukezic, A., Berg, A., et al., 2019. The seventh visual object tracking vot2019 challenge results. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops. 0\u20130."},{"key":"10.1016\/j.engappai.2026.115425_b28","unstructured":"Kristan, M., Matas, J., Leonardis, A., Felsberg, M., Pflugfelder, R., K\u00e4m\u00e4r\u00e4inen, J.-K., Chang, H.J., Danelljan, M., Cehovin, L., Luke\u017ei\u010d, A., et al., 2021. The ninth visual object tracking vot2021 challenge results. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 2711\u20132738."},{"key":"10.1016\/j.engappai.2026.115425_b29","doi-asserted-by":"crossref","unstructured":"Lai, Z., Lu, E., Xie, W., 2020. Mast: A memory-augmented self-supervised tracker. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6479\u20136488.","DOI":"10.1109\/CVPR42600.2020.00651"},{"key":"10.1016\/j.engappai.2026.115425_b30","doi-asserted-by":"crossref","unstructured":"Li, P., Chen, B., Ouyang, W., Wang, D., Yang, X., Lu, H., 2019. Gradnet: Gradient-guided network for visual object tracking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6162\u20136171.","DOI":"10.1109\/ICCV.2019.00626"},{"key":"10.1016\/j.engappai.2026.115425_b31","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, W., Wang, Q., Zhang, F., Xing, J., Yan, J., 2019. Siamrpn++: Evolution of siamese visual tracking with very deep networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4282\u20134291.","DOI":"10.1109\/CVPR.2019.00441"},{"key":"10.1016\/j.engappai.2026.115425_b32","doi-asserted-by":"crossref","unstructured":"Li, B., Yan, J., Wu, W., Zhu, Z., Hu, X., 2018. High performance visual tracking with siamese region proposal network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 8971\u20138980.","DOI":"10.1109\/CVPR.2018.00935"},{"key":"10.1016\/j.engappai.2026.115425_b33","series-title":"Siamvgg: Visual tracking using deeper siamese networks","author":"Li","year":"2019"},{"key":"10.1016\/j.engappai.2026.115425_b34","series-title":"European Conference on Computer Vision","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.engappai.2026.115425_b35","article-title":"Tracking with mutual attention network","author":"Liu","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2026.115425_b36","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.engappai.2026.115425_b37","doi-asserted-by":"crossref","unstructured":"Liu, X., Zhou, L., Zhou, Z., Chen, J., He, Z., 2025. Mambavlt: Time-evolving multimodal state space model for vision-language tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8731\u20138741.","DOI":"10.1109\/CVPR52734.2025.00816"},{"key":"10.1016\/j.engappai.2026.115425_b38","series-title":"European Conference on Computer Vision","first-page":"445","article-title":"A benchmark and simulator for uav tracking","author":"Mueller","year":"2016"},{"key":"10.1016\/j.engappai.2026.115425_b39","doi-asserted-by":"crossref","unstructured":"Muller, M., Bibi, A., Giancola, S., Alsubaihi, S., Ghanem, B., 2018. Trackingnet: A large-scale dataset and benchmark for object tracking in the wild. In: Proceedings of the European Conference on Computer Vision. ECCV, pp. 300\u2013317.","DOI":"10.1007\/978-3-030-01246-5_19"},{"key":"10.1016\/j.engappai.2026.115425_b40","doi-asserted-by":"crossref","unstructured":"Nam, H., Han, B., 2016. Learning multi-domain convolutional neural networks for visual tracking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4293\u20134302.","DOI":"10.1109\/CVPR.2016.465"},{"key":"10.1016\/j.engappai.2026.115425_b41","doi-asserted-by":"crossref","unstructured":"Oh, S.W., Lee, J.-Y., Xu, N., Kim, S.J., 2019. Video object segmentation using space\u2013time memory networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9226\u20139235.","DOI":"10.1109\/ICCV.2019.00932"},{"key":"10.1016\/j.engappai.2026.115425_b42","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.108682","article-title":"Unified spatio-temporal attention mixformer for visual object tracking","volume":"134","author":"Park","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2026.115425_b43","article-title":"Dual adversity training for domain generalization","author":"Shao","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.engappai.2026.115425_b44","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107757","article-title":"Comprehensive disentanglement with fine-grained feature mitigation for domain generalization","volume":"191","author":"Shao","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.engappai.2026.115425_b45","doi-asserted-by":"crossref","unstructured":"Sun, C., Wang, D., Lu, H., Yang, M.-H., 2018a. Correlation tracking via joint discrimination and reliability learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 489\u2013497.","DOI":"10.1109\/CVPR.2018.00058"},{"key":"10.1016\/j.engappai.2026.115425_b46","doi-asserted-by":"crossref","unstructured":"Sun, C., Wang, D., Lu, H., Yang, M.-H., 2018b. Learning spatial-aware regressions for visual tracking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 8962\u20138970.","DOI":"10.1109\/CVPR.2018.00934"},{"key":"10.1016\/j.engappai.2026.115425_b47","doi-asserted-by":"crossref","unstructured":"Valmadre, J., Bertinetto, L., Henriques, J., Vedaldi, A., Torr, P.H., 2017. End-to-end representation learning for correlation filter based tracking. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 2805\u20132813.","DOI":"10.1109\/CVPR.2017.531"},{"key":"10.1016\/j.engappai.2026.115425_b48","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.115425_b49","doi-asserted-by":"crossref","unstructured":"Videnovic, Jovana, Lukezic, Alan, Kristan, Matej, 2025. A distractor-aware memory for visual object tracking with sam2. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 24255\u201324264.","DOI":"10.1109\/CVPR52734.2025.02259"},{"key":"10.1016\/j.engappai.2026.115425_b50","doi-asserted-by":"crossref","unstructured":"Wang, C., Fang, X., Tiwari, P., 2025a. DyPolySeg: Taylor series-inspired dynamic polynomial fitting network for few-shot point cloud semantic segmentation. In: Proceedings of the Forty-Second International Conference on Machine Learning.","DOI":"10.1609\/aaai.v39i7.32810"},{"key":"10.1016\/j.engappai.2026.115425_b51","doi-asserted-by":"crossref","unstructured":"Wang, C., He, S., Fang, X., et al., 2025b. Point clouds meets physics: Dynamic acoustic field fitting network for point cloud understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 22182\u201322192.","DOI":"10.1109\/CVPR52734.2025.02066"},{"key":"10.1016\/j.engappai.2026.115425_b52","unstructured":"Wang, C., He, S., Fang, X., et al., 2025c. Reasoning beyond points: A visual introspective approach for few-shot 3D segmentation. In: Proceedings of the Thirty-Ninth Annual Conference on Neural Information Processing Systems."},{"key":"10.1016\/j.engappai.2026.115425_b53","doi-asserted-by":"crossref","unstructured":"Wang, G., Luo, C., Xiong, Z., Zeng, W., 2019. Spm-tracker: Series-parallel matching for real-time visual object tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3643\u20133652.","DOI":"10.1109\/CVPR.2019.00376"},{"key":"10.1016\/j.engappai.2026.115425_b54","doi-asserted-by":"crossref","unstructured":"Wang, Q., Zhang, L., Bertinetto, L., Hu, W., Torr, P.H., 2019. Fast online object tracking and segmentation: A unifying approach. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1328\u20131338.","DOI":"10.1109\/CVPR.2019.00142"},{"key":"10.1016\/j.engappai.2026.115425_b55","doi-asserted-by":"crossref","unstructured":"Wang, N., Zhou, W., Wang, J., Li, H., 2021. Transformer meets tracker: Exploiting temporal context for robust visual tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1571\u20131580.","DOI":"10.1109\/CVPR46437.2021.00162"},{"key":"10.1016\/j.engappai.2026.115425_b56","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S., 2018. Cbam: Convolutional block attention module. In: Proceedings of the European Conference on Computer Vision. pp. 3\u201319.","DOI":"10.1007\/978-3-030-01234-2_1"},{"issue":"9","key":"10.1016\/j.engappai.2026.115425_b57","doi-asserted-by":"crossref","first-page":"1834","DOI":"10.1109\/TPAMI.2014.2388226","article-title":"Object tracking benchmark","volume":"37","author":"Wu","year":"2015","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2026.115425_b58","doi-asserted-by":"crossref","unstructured":"Wu, Y., Wang, X., Yang, X., Liu, M., Zeng, D., Ye, H., Li, S., 2025. Learning Occlusion-Robust Vision Transformers for Real-Time UAV Tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 17103\u201317113.","DOI":"10.1109\/CVPR52734.2025.01594"},{"key":"10.1016\/j.engappai.2026.115425_b59","doi-asserted-by":"crossref","first-page":"2791","DOI":"10.1109\/TMM.2021.3087340","article-title":"Learning temporal-correlated and channel-decorrelated siamese networks for visual tracking","volume":"24","author":"Xi","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2026.115425_b60","doi-asserted-by":"crossref","unstructured":"Xie, J., Zhong, B., Zhang, S., Shi, L., Song, S., Ji, R., 2024. Autoregressive queries for adaptive tracking with spatio-temporal transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 19300\u201319309.","DOI":"10.1109\/CVPR52733.2024.01826"},{"key":"10.1016\/j.engappai.2026.115425_b61","doi-asserted-by":"crossref","unstructured":"Yan, B., Peng, H., Fu, J., Wang, D., Lu, H., 2021a. Learning spatio-temporal transformer for visual tracking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10448\u201310457.","DOI":"10.1109\/ICCV48922.2021.01028"},{"key":"10.1016\/j.engappai.2026.115425_b62","doi-asserted-by":"crossref","unstructured":"Yan, B., Zhang, X., Wang, D., Lu, H., Yang, X., 2021. Alpha-refine: Boosting tracking performance by precise bounding box estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 5289\u20135298.","DOI":"10.1109\/CVPR46437.2021.00525"},{"key":"10.1016\/j.engappai.2026.115425_b63","doi-asserted-by":"crossref","unstructured":"Yang, T., Chan, A.B., 2018b. Learning dynamic memory networks for object tracking. In: Proceedings of the European Conference on Computer Vision. pp. 152\u2013167.","DOI":"10.1007\/978-3-030-01240-3_10"},{"key":"10.1016\/j.engappai.2026.115425_b64","doi-asserted-by":"crossref","first-page":"1956","DOI":"10.1109\/TMM.2021.3074239","article-title":"Siamcorners: Siamese corner networks for visual tracking","volume":"24","author":"Yang","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2026.115425_b65","doi-asserted-by":"crossref","unstructured":"Zhang, L., Gonzalez-Garcia, A., Weijer, J.v.d., Danelljan, M., Khan, F.S., 2019. Learning the model update for siamese trackers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 4010\u20134019.","DOI":"10.1109\/ICCV.2019.00411"},{"key":"10.1016\/j.engappai.2026.115425_b66","series-title":"Trtr: Visual tracking with transformer","author":"Zhao","year":"2021"},{"key":"10.1016\/j.engappai.2026.115425_b67","doi-asserted-by":"crossref","first-page":"2098","DOI":"10.1109\/TMM.2021.3075876","article-title":"Bilateral weighted regression ranking model with spatial\u2013temporal correlation filter for visual tracking","volume":"24","author":"Zhu","year":"2021","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2026.115425_b68","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Wang, Q., Li, B., Wu, W., Yan, J., Hu, W., 2018. Distractor-aware siamese networks for visual object tracking. In: Proceedings of the European Conference on Computer Vision. pp. 101\u2013117.","DOI":"10.1007\/978-3-030-01240-3_7"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626017094?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626017094?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T13:54:10Z","timestamp":1782482050000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626017094"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":68,"alternative-id":["S0952197626017094"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115425","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Hierarchical Spatial\u2013Temporal Feature Integration Network for visual tracking","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115425","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"115425"}}