{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:07:14Z","timestamp":1757617634025,"version":"3.44.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T00:00:00Z","timestamp":1743033600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T00:00:00Z","timestamp":1743033600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s00371-025-03861-5","type":"journal-article","created":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T11:52:38Z","timestamp":1743335558000},"page":"8153-8167","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing multi-object tracking efficiency through result-guided feature extraction and query filtering"],"prefix":"10.1007","volume":"41","author":[{"given":"Yan","family":"Ding","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yushen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingfeng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bozhi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaxin","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingxi","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhe","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weidong","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,27]]},"reference":[{"issue":"2","key":"3861_CR1","doi-asserted-by":"publisher","first-page":"385","DOI":"10.1007\/s00779-019-01296-z","volume":"26","author":"H Ahn","year":"2022","unstructured":"Ahn, H., Cho, H.J.: Research of multi-object detection and tracking using machine learning based on knowledge for video surveillance system. Pers. Ubiquit. Comput. 26(2), 385\u2013394 (2022)","journal-title":"Pers. Ubiquit. Comput."},{"doi-asserted-by":"crossref","unstructured":"Cheng, C. C., Qiu, M. X., Chiang, C. K., et al. Rest: a reconfigurable spatial-temporal graph model for multi-camera multi-object tracking, Proceedings of the IEEE\/CVF International Conference on Computer Vision, 10051\u201310060 (2023).","key":"3861_CR2","DOI":"10.1109\/ICCV51070.2023.00922"},{"doi-asserted-by":"crossref","unstructured":"Li, P., Jin, J.: Time3d: end-to-end joint monocular 3d object detection and tracking for autonomous driving[C]\/\/Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 3885\u20133894 (2022).","key":"3861_CR3","DOI":"10.1109\/CVPR52688.2022.00386"},{"issue":"12","key":"3861_CR4","doi-asserted-by":"publisher","first-page":"3300","DOI":"10.1049\/ipr2.12565","volume":"16","author":"Y Xue","year":"2022","unstructured":"Xue, Y., Jin, G., Shen, T., et al.: MobileTrack: Siamese efficient mobile network for high-speed UAV tracking. IET Image Proc. 16(12), 3300\u20133313 (2022)","journal-title":"IET Image Proc."},{"doi-asserted-by":"crossref","unstructured":"Liu, S., Li, X., Lu, H., et al.: Multi-object tracking meets moving UAV, Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8876\u20138885 (2022).","key":"3861_CR5","DOI":"10.1109\/CVPR52688.2022.00867"},{"issue":"9","key":"3861_CR6","doi-asserted-by":"publisher","first-page":"299","DOI":"10.1016\/j.cja.2023.03.048","volume":"36","author":"Y Xue","year":"2023","unstructured":"Xue, Y., Jin, G., Shen, T., et al.: Template-guided frequency attention and adaptive cross-entropy loss for UAV visual tracking. Chin. J. Aeronaut. 36(9), 299\u2013312 (2023)","journal-title":"Chin. J. Aeronaut."},{"issue":"1","key":"3861_CR7","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s00371-023-02783-4","volume":"40","author":"M Yang","year":"2024","unstructured":"Yang, M., Feng, Y., Rao, A.S., et al.: Evolving graph-based video crowd anomaly detection. Vis. Comput. 40(1), 303\u2013318 (2024)","journal-title":"Vis. Comput."},{"doi-asserted-by":"crossref","unstructured":"Li, H., Yang, M., Yang, C., et al.: Soccer match broadcast video analysis method based on detection and tracking. Computer Animation and Virtual Worlds, 35(3): e2259: 1\u201317 (2024).","key":"3861_CR8","DOI":"10.1002\/cav.2259"},{"doi-asserted-by":"crossref","unstructured":"Wojke, N., Bewley, A., Paulus, D.: Simple online and realtime tracking with a deep association metric, 2017 IEEE international conference on image processing (ICIP). IEEE, 3645\u20133649 (2017).","key":"3861_CR9","DOI":"10.1109\/ICIP.2017.8296962"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Y., Sun, P., Jiang, Y. et al.: Bytetrack: multi-object tracking by associating every detection box, European conference on computer vision. Cham: Springer Nature Switzerland, 1\u201321 (2022).","key":"3861_CR10","DOI":"10.1007\/978-3-031-20047-2_1"},{"unstructured":"Aharon, N., Orfaig, R., Bobrovsky, B. Z.: BoT-SORT: Robust associations multi-pedestrian tracking. arXiv preprint arXiv:2206.14651, (2022).","key":"3861_CR11"},{"doi-asserted-by":"crossref","unstructured":"Wang, Z., Zheng, L., Liu, Y., et al.: Towards real-time multi-object tracking, European conference on computer vision. Cham: Springer International Publishing, 107\u2013122 (2020).","key":"3861_CR12","DOI":"10.1007\/978-3-030-58621-8_7"},{"key":"3861_CR13","doi-asserted-by":"publisher","first-page":"3069","DOI":"10.1007\/s11263-021-01513-4","volume":"129","author":"Y Zhang","year":"2021","unstructured":"Zhang, Y., Wang, C., Wang, X., et al.: Fairmot: on the fairness of detection and re-identification in multiple object tracking. Int. J. Comput. Vision 129, 3069\u20133087 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"3861_CR14","doi-asserted-by":"publisher","first-page":"3182","DOI":"10.1109\/TIP.2022.3165376","volume":"31","author":"C Liang","year":"2022","unstructured":"Liang, C., Zhang, Z., Zhou, X., et al.: Rethinking the competition between detection and reid in multi object tracking. IEEE Trans. Image Process. 31, 3182\u20133196 (2022)","journal-title":"IEEE Trans. Image Process."},{"unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al.: Attention is all you need. arXiv preprint arXiv:1706.03762, (2017).","key":"3861_CR15"},{"unstructured":"Dosovitskiy, A. An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929, (2020).","key":"3861_CR16"},{"doi-asserted-by":"crossref","unstructured":"Chen, C. F. R., Fan, Q., Panda, R.: Crossvit: cross-attention multi-scale vision transformer for image classification, Proceedings of the IEEE\/CVF international conference on computer vision. 357\u2013366 (2021).","key":"3861_CR17","DOI":"10.1109\/ICCV48922.2021.00041"},{"doi-asserted-by":"crossref","unstructured":"He, S., Luo, H., Wang, P., et al.: Transreid: transformer-based object re-identification, Proceedings of the IEEE\/CVF international conference on computer vision. 15013\u201315022 (2021).","key":"3861_CR18","DOI":"10.1109\/ICCV48922.2021.01474"},{"issue":"9","key":"3861_CR19","doi-asserted-by":"publisher","first-page":"4087","DOI":"10.1007\/s00371-022-02577-0","volume":"39","author":"N Pervaiz","year":"2023","unstructured":"Pervaiz, N., Fraz, M.M., Shahzad, M.: Per-former: rethinking person re-identification using transformer augmented with self-attention and contextual mapping. Vis. Comput. 39(9), 4087\u20134102 (2023)","journal-title":"Vis. Comput."},{"doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., et al.: End-to-end object detection with transformers, European conference on computer vision. Cham: Springer International Publishing, 213\u2013229 (2020).","key":"3861_CR20","DOI":"10.1007\/978-3-030-58452-8_13"},{"unstructured":"Zhu, X., Su, W., Lu, L. et al.: Deformable detr: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159, (2020).","key":"3861_CR21"},{"unstructured":"Liu, S., Li, F., Zhang, H., et al.: Dab-detr: dynamic anchor boxes are better queries for detr. arXiv preprint arXiv:2201.12329, (2022).","key":"3861_CR22"},{"doi-asserted-by":"crossref","unstructured":"Xue, Y., Jin, G., Shen, T. et al.: Smalltrack: wavelet pooling and graph enhanced classification for uav small object tracking. IEEE Transactions on Geoscience and Remote Sensing, (2023).","key":"3861_CR23","DOI":"10.1109\/TGRS.2023.3305728"},{"issue":"11","key":"3861_CR24","doi-asserted-by":"publisher","first-page":"6571","DOI":"10.1109\/TCSVT.2023.3263884","volume":"33","author":"M Hu","year":"2023","unstructured":"Hu, M., Zhu, X., Wang, H., et al.: Stdformer: spatial-temporal motion transformer for multiple object tracking. IEEE Trans. Circuits Syst. Video Technol. 33(11), 6571\u20136594 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"doi-asserted-by":"crossref","unstructured":"Xue, Y., Jin, G., Shen, T. et al.: Consistent representation mining for multi-drone single object tracking. IEEE Transactions on Circuits and Systems for Video Technology, (2024).","key":"3861_CR25","DOI":"10.1109\/TCSVT.2024.3411301"},{"unstructured":"Gao, R., Zhang, Y., Wang, L.: Multiple object tracking as ID prediction. arXiv preprint arXiv:2403.16848, (2024).","key":"3861_CR26"},{"doi-asserted-by":"crossref","unstructured":"Xue, Y., Shen, T., Jin, G. et al.: Handling occlusion in uav visual tracking with query-guided redetection. IEEE Transactions on Instrumentation and Measurement, (2024).","key":"3861_CR27","DOI":"10.1109\/TIM.2024.3440378"},{"doi-asserted-by":"crossref","unstructured":"Meinhardt, T., Kirillov, A., Leal-Taixe, L. et al.: Trackformer: multi-object tracking with transformers, Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 8844\u20138854 (2022).","key":"3861_CR28","DOI":"10.1109\/CVPR52688.2022.00864"},{"key":"3861_CR29","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TMM.2021.3120873","volume":"25","author":"X Lin","year":"2021","unstructured":"Lin, X., Sun, S., Huang, W., et al.: EAPT: efficient attention pyramid transformer for image processing. IEEE Trans. Multimedia 25, 50\u201361 (2021)","journal-title":"IEEE Trans. Multimedia"},{"doi-asserted-by":"crossref","unstructured":"Zeng, F., Dong, B., Zhang, Y., et al.: Motr: end-to-end multiple-object tracking with transformer, European Conference on Computer Vision. Cham: Springer Nature Switzerland, 659\u2013675 (2022).","key":"3861_CR30","DOI":"10.1007\/978-3-031-19812-0_38"},{"key":"3861_CR31","doi-asserted-by":"publisher","first-page":"548","DOI":"10.1007\/s11263-020-01375-2","volume":"129","author":"J Luiten","year":"2021","unstructured":"Luiten, J., Osep, A., Dendorfer, P., Torr, P., Geiger, A., Leal-Taix\u00e9, L., Leibe, B.: Hota: a higher order metric for evaluating multi-object tracking. Int. J. Comput. Vis. 129, 548\u201378 (2021). https:\/\/doi.org\/10.1007\/s11263-020-01375-2","journal-title":"Int. J. Comput. Vis."},{"key":"3861_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2008\/246309","volume":"2008","author":"K Bernardin","year":"2008","unstructured":"Bernardin, K., Stiefelhagen, R.: Evaluating multiple object tracking performance: the clear mot metrics. EURASIP J. Image Video Process. 2008, 1\u201310 (2008)","journal-title":"EURASIP J. Image Video Process."},{"doi-asserted-by":"crossref","unstructured":"Ristani, E., Solera, F., Zou, R. et al.: Performance measures and a data set for multi-target, multi-camera tracking, European conference on computer vision. Cham: Springer International Publishing, 17\u201335 (2016).","key":"3861_CR33","DOI":"10.1007\/978-3-319-48881-3_2"},{"doi-asserted-by":"crossref","unstructured":"Bewley, A., Ge, Z., Ott, L, et al.: Simple online and realtime tracking, 2016 IEEE international conference on image processing (ICIP). IEEE, 3464\u20133468 (2016).","key":"3861_CR34","DOI":"10.1109\/ICIP.2016.7533003"},{"unstructured":"Ge, Z., Liu, S., Wang, F. et al.: Yolox: exceeding yolo series in 2021. arXiv preprint arXiv:2107.08430, (2021).","key":"3861_CR35"},{"doi-asserted-by":"crossref","unstructured":"Cao, J., Pang, J., Weng, X. et al.: Observation-centric sort: Rethinking sort for robust multi-object tracking, Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 9686\u20139696 (2023).","key":"3861_CR36","DOI":"10.1109\/CVPR52729.2023.00934"},{"unstructured":"Farhadi, A., Redmon, J.: Yolov3: an incremental improvement, Computer vision and pattern recognition. Berlin\/Heidelberg, Germany: Springer, 2018, 1804: 1\u20136","key":"3861_CR37"},{"doi-asserted-by":"crossref","unstructured":"Yu, F., Wang, D., Shelhamer, E. et al.: Deep layer aggregation, Proceedings of the IEEE conference on computer vision and pattern recognition. 2403\u20132412 (2018).","key":"3861_CR38","DOI":"10.1109\/CVPR.2018.00255"},{"doi-asserted-by":"crossref","unstructured":"Zhou, X., Koltun, V., Kr\u00e4henb\u00fchl, P.: Tracking objects as points, European conference on computer vision. Cham: Springer International Publishing, 474\u2013490 (2020).","key":"3861_CR39","DOI":"10.1007\/978-3-030-58548-8_28"},{"doi-asserted-by":"crossref","unstructured":"Zhou, X., Yin, T., Koltun, V., et al.: Global tracking transformers, Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 8771\u20138780 (2022).","key":"3861_CR40","DOI":"10.1109\/CVPR52688.2022.00857"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Y., Wang, T., Zhang, X.: Motrv2: bootstrapping end-to-end multi-object tracking by pretrained object detectors, Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 22056\u201322065 (2023).","key":"3861_CR41","DOI":"10.1109\/CVPR52729.2023.02112"},{"doi-asserted-by":"crossref","unstructured":"Gao, R., Wang, L.: MeMOTR: long-term memory-augmented transformer for multi-object tracking, Proceedings of the IEEE\/CVF International Conference on Computer Vision. 9901\u20139910 (2023).","key":"3861_CR42","DOI":"10.1109\/ICCV51070.2023.00908"},{"unstructured":"Yan, F., Luo, W., Zhong, Y. et al.: Bridging the gap between end-to-end and non-end-to-end multi-object tracking. arXiv preprint arXiv:2305.12724, (2023).","key":"3861_CR43"},{"unstructured":"Zhao, Y., Lv, W., Xu, S., et al.: DETRs beat YOLOs on real-time object detection. arXiv e-prints: arXiv: 2304.08069 (2023).","key":"3861_CR44"},{"doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S. et al.: Deep residual learning for image recognition, Proceedings of the IEEE conference on computer vision and pattern recognition. 770\u2013778 (2016).","key":"3861_CR45","DOI":"10.1109\/CVPR.2016.90"},{"issue":"4","key":"3861_CR46","doi-asserted-by":"publisher","first-page":"1075","DOI":"10.1007\/s11263-023-01922-7","volume":"132","author":"S Hao","year":"2024","unstructured":"Hao, S., Liu, P., Zhan, Y., et al.: Divotrack: a novel dataset and baseline method for cross-view multi-object tracking in diverse open scenes. Int. J. Comput. Vision 132(4), 1075\u20131090 (2024)","journal-title":"Int. J. Comput. Vision"},{"unstructured":"Milan, A.: MOT16: a benchmark for multi-object tracking. arXiv preprint arXiv:1603.00831, (2016).","key":"3861_CR47"},{"unstructured":"Shao, S., Zhao, Z., Li, B. et al. Crowdhuman: a benchmark for detecting human in a crowd. arXiv preprint arXiv:1805.00123, (2018).","key":"3861_CR48"},{"doi-asserted-by":"crossref","unstructured":"Bergmann, P., Meinhardt, T., Leal-Taixe, L.: Tracking without bells and whistles, Proceedings of the IEEE\/CVF international conference on computer vision. 941\u2013951 (2019).","key":"3861_CR49","DOI":"10.1109\/ICCV.2019.00103"},{"doi-asserted-by":"crossref","unstructured":"Wu, J., Cao, J., Song, L. et al.: Track to detect and segment: an online multi-object tracker, Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 12352\u201312361 (2021).","key":"3861_CR50","DOI":"10.1109\/CVPR46437.2021.01217"},{"unstructured":"He, L., Liao, X., Liu, W. et al. FastReID: a pytorch toolbox for general instance re-identification. arXiv preprint arXiv:2006.02631, (2020).","key":"3861_CR51"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03861-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03861-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03861-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T08:26:51Z","timestamp":1757147211000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03861-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,27]]},"references-count":51,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["3861"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03861-5","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2025,3,27]]},"assertion":[{"value":"24 February 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 March 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}