{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T19:37:42Z","timestamp":1782934662782,"version":"3.54.5"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004479","name":"Jiangxi Provincial Natural Science Foundation","doi-asserted-by":"crossref","award":["20242BAB25058"],"award-info":[{"award-number":["20242BAB25058"]}],"id":[{"id":"10.13039\/501100004479","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11760-026-05463-7","type":"journal-article","created":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T19:51:27Z","timestamp":1780948287000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Transformer visual tracking via cascaded attention mechanisms"],"prefix":"10.1007","volume":"20","author":[{"given":"Yuanyun","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingzhen","family":"Si","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jilong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Geng","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,8]]},"reference":[{"issue":"11","key":"5463_CR1","doi-asserted-by":"publisher","first-page":"10845","DOI":"10.1109\/TCSVT.2024.3411301","volume":"34","author":"Y Xue","year":"2024","unstructured":"Xue, Y., Jin, G., Shen, T., Tan, L., Wang, N., Gao, J., Wang, L.: Consistent representation mining for multi-drone single object tracking. IEEE Trans. Circuits Syst. Video Technol. 34(11), 10845\u201310859 (2024)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"5463_CR2","first-page":"1","volume":"61","author":"Y Xue","year":"2023","unstructured":"Xue, Y., Jin, G., Shen, T., Tan, L., Wang, N., Gao, J., Wang, L.: Smalltrack: Wavelet pooling and graph enhanced classification for uav small object tracking. IEEE Trans. Geosci. Remote Sens. 61, 1\u201315 (2023)","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"5463_CR3","doi-asserted-by":"crossref","unstructured":"Cui, Y., Jiang, C., Wang, L., Wu, G.: Mixformer: End-to-end tracking with iterative mixed attention, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (2022), pp. 13608\u201313618","DOI":"10.1109\/CVPR52688.2022.01324"},{"key":"5463_CR4","doi-asserted-by":"crossref","unstructured":"Ye, B., Chang, H., Ma, B., Shan, S., Chen, X.: Joint feature learning and relation modeling for tracking: A one-stream framework, in: European Conference on Computer Vision, Springer, (2022), pp. 341\u2013357","DOI":"10.1007\/978-3-031-20047-2_20"},{"issue":"8","key":"5463_CR5","doi-asserted-by":"publisher","first-page":"7554","DOI":"10.1109\/TCSVT.2025.3549953","volume":"35","author":"Y Xue","year":"2025","unstructured":"Xue, Y., Zhong, B., Jin, G., Shen, T., Tan, L., Li, N.: Avltrack: Dynamic sparse learning for aerial vision-language tracking. IEEE Trans. Circuits Syst. Video Technol. 35(8), 7554\u20137567 (2025)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"5463_CR6","doi-asserted-by":"crossref","unstructured":"Wang, N., Zhou, W., Wang, J., Li, H.: Transformer meets tracker: Exploiting temporal context for robust visual tracking, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2021), pp. 1571\u20131580","DOI":"10.1109\/CVPR46437.2021.00162"},{"key":"5463_CR7","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.\u00a0N.,\u00a0Kaiser, \u0141., Polosukhin, I.: Attention is all you need, Advances in neural information processing systems 30 (2017)"},{"key":"5463_CR8","doi-asserted-by":"crossref","unstructured":"Bertinetto, L., Valmadre, J., Henriques, J.\u00a0F., Vedaldi, A., Torr, P.\u00a0H.: Fully-convolutional siamese networks for object tracking, in: European conference on computer, (2016), pp. 850\u2013865","DOI":"10.1007\/978-3-319-48881-3_56"},{"key":"5463_CR9","doi-asserted-by":"crossref","unstructured":"Dong, X., Shen, J.: Triplet loss in siamese network for object tracking, in: Proceedings of the European conference on computer vision (ECCV), (2018), pp. 459\u2013474","DOI":"10.1007\/978-3-030-01261-8_28"},{"key":"5463_CR10","unstructured":"Yan, B., Peng, H., Fu, J., Wang, D., Lu, H.: Learning spatio-temporal transformer for visual tracking, in: Proceedings of the IEEE\/CVF international conference on computer vision, (2021), pp. 10448\u201310457"},{"key":"5463_CR11","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers, in: European conference on computer vision, Springer, (2020), pp. 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"5463_CR12","doi-asserted-by":"crossref","unstructured":"Chen, X., Yan, B., Zhu, J., Wang, D., Yang, X., Lu, H.: Transformer tracking, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2021), pp. 8126\u20138135","DOI":"10.1109\/CVPR46437.2021.00803"},{"key":"5463_CR13","doi-asserted-by":"crossref","unstructured":"Fu, Z., Liu, Q., Fu, Z., Wang, Y.: Stmtrack: Template-free visual tracking with space-time memory networks, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2021), pp. 13774\u201313783","DOI":"10.1109\/CVPR46437.2021.01356"},{"key":"5463_CR14","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.130571","volume":"647","author":"N Wang","year":"2025","unstructured":"Wang, N., Cui, Z., Li, A., Xue, Y., Wang, R., Nie, F.: Multi-order graph based clustering via dynamical low rank tensor approximation. Neurocomputing 647, 130571 (2025)","journal-title":"Neurocomputing"},{"key":"5463_CR15","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S.: An image is worth 16x16 words: Transformers for image recognition at scale, (2020). arXiv:2010.11929 arXiv preprint"},{"key":"5463_CR16","doi-asserted-by":"crossref","unstructured":"Chollet, F.: Xception: Deep learning with depthwise separable convolutions, in: Proceedings of the IEEE conference on computer vision and pattern recognition, (2017), pp. 1251\u20131258","DOI":"10.1109\/CVPR.2017.195"},{"key":"5463_CR17","doi-asserted-by":"crossref","unstructured":"Fan, H., Lin, L., Yang, F., Chu, P., Deng, G., Yu, S., Bai, H., Xu, Y., Liao, C., Ling, H.: Lasot: A high-quality benchmark for large-scale single object tracking, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2019), pp. 5374\u20135383","DOI":"10.1109\/CVPR.2019.00552"},{"issue":"5","key":"5463_CR18","doi-asserted-by":"publisher","first-page":"1562","DOI":"10.1109\/TPAMI.2019.2957464","volume":"43","author":"L Huang","year":"2019","unstructured":"Huang, L., Zhao, X., Huang, K.: Got-10k: A large high-diversity benchmark for generic object tracking in the wild. IEEE Trans. Pattern Anal. Mach. Intell. 43(5), 1562\u20131577 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5463_CR19","doi-asserted-by":"crossref","unstructured":"Muller, M., Bibi, A., Giancola, S., Alsubaihi, S., Ghanem, B.: Trackingnet: A large-scale dataset and benchmark for object tracking in the wild, in: Proceedings of the European conference on computer vision, (2018), pp. 300\u2013317","DOI":"10.1007\/978-3-030-01246-5_19"},{"key":"5463_CR20","doi-asserted-by":"crossref","unstructured":"Mueller, M., Smith, N., Ghanem, B.: A benchmark and simulator for uav tracking, in: European conference on computer vision, (2016), pp. 445\u2013461","DOI":"10.1007\/978-3-319-46448-0_27"},{"issue":"11","key":"5463_CR21","doi-asserted-by":"publisher","first-page":"11362","DOI":"10.1109\/TCSVT.2025.3578479","volume":"35","author":"J Wang","year":"2025","unstructured":"Wang, J., Chai, B., Zhou, L., Wang, Y.: Robust object tracking via long-range spatial representation and local feature enhancement. IEEE Trans. Circuits Syst. Video Technol. 35(11), 11362\u201311376 (2025)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"5463_CR22","unstructured":"Wang, J., Yin, P., Wang, Y., Yang, W.: Cmat: Integrating convolution mixer and self-attention for visual tracking, IEEE Transactions on Multimedia (2023)"},{"key":"5463_CR23","doi-asserted-by":"crossref","unstructured":"Yang, K., Zhang, H., Shi, J., Ma, J.: Bandt: A border-aware network with deformable transformers for visual tracking. IEEE Trans. Consum. Electron. , (2023)","DOI":"10.1109\/TCE.2023.3251407"},{"key":"5463_CR24","unstructured":"Wang, J., Lai, C., Zhang, W., Wang, Y., Meng, C.: Transformer tracking with multi-scale dual-attention, Complex & Intelligent Systems (2023) 1\u201314"},{"key":"5463_CR25","doi-asserted-by":"crossref","unstructured":"Blatter, P., Kanakis, M., Danelljan, M., Van\u00a0Gool, L.: Efficient visual tracking with exemplar transformers, in: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, (2023), pp. 1571\u20131581","DOI":"10.1109\/WACV56688.2023.00162"},{"key":"5463_CR26","unstructured":"Zhu, J., Chen, X., Wang, D., Zhao, W., Lu, H.: Srrt: Search region regulation tracking, (2022). arXiv:2207.04438 arXiv preprint"},{"key":"5463_CR27","doi-asserted-by":"crossref","unstructured":"Ma, F., Shou, M.\u00a0Z., Zhu, L., Fan, H., Xu, Y., Yang, Y., Yan, Z.: Unified transformer tracker for object tracking, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (2022), pp. 8781\u20138790","DOI":"10.1109\/CVPR52688.2022.00858"},{"key":"5463_CR28","doi-asserted-by":"crossref","unstructured":"Xie, F., Wang, C., Wang, G., Cao, Y., Yang, W., Zeng, W.: Correlation-aware deep tracking, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (2022), pp. 8751\u20138760","DOI":"10.1109\/CVPR52688.2022.00855"},{"key":"5463_CR29","unstructured":"Zhao, M., Okada, K., Inaba, M.: Trtr: Visual tracking with transformer, (2021). arXiv:2105.03817 arXiv preprint"},{"key":"5463_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Liu, Y., Wang, X., Li, B., Hu, W.: Learn to match: Automatic matching network design for visual tracking, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, (2021), pp. 13339\u201313348","DOI":"10.1109\/ICCV48922.2021.01309"},{"key":"5463_CR31","doi-asserted-by":"crossref","unstructured":"Cao, Z., Fu, C., Ye, J., Li, B., Li, Y.: Hift: Hierarchical feature transformer for aerial tracking, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, (2021), pp. 15457\u201315466","DOI":"10.1109\/ICCV48922.2021.01517"},{"key":"5463_CR32","unstructured":"Cui, Y., Jiang, C., Wang, L., Wu, G.: Target transformed regression for accurate tracking, (2021). arXiv:2104.00403 arXiv preprint"},{"key":"5463_CR33","unstructured":"Kristan, M., Leonardis, A., Matas, J., Felsberg, M., Pflugfelder, R., K\u00e4m\u00e4r\u00e4inen, J.-K., Danelljan, M., \u010c. Zajc, L., Luke\u017ei\u010d, A.: et\u00a0al., The eighth visual object tracking vot2020 challenge results, in: Proceedings of the European conference on computer vision, (2020), pp. 547\u2013601"},{"key":"5463_CR34","unstructured":"Kiani\u00a0Galoogahi, H., Fagg, A., Huang, C., Ramanan, D., Lucey, S.: Need for speed: A benchmark for higher frame rate object tracking, in: Proceedings of the IEEE International Conference on Computer Vision, (2017), pp. 1125\u20131134"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05463-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-026-05463-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05463-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T17:49:52Z","timestamp":1782928192000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-026-05463-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":34,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["5463"],"URL":"https:\/\/doi.org\/10.1007\/s11760-026-05463-7","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"13 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 April 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 May 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 June 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"402"}}