{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T17:02:19Z","timestamp":1783530139433,"version":"3.55.0"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,4,8]],"date-time":"2024-04-08T00:00:00Z","timestamp":1712534400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,4,8]],"date-time":"2024-04-08T00:00:00Z","timestamp":1712534400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62203012"],"award-info":[{"award-number":["62203012"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61901221"],"award-info":[{"award-number":["61901221"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61871123"],"award-info":[{"award-number":["61871123"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Natural Science Foundation of the Anhui Higher Education Institutions of China","award":["2023AH030020"],"award-info":[{"award-number":["2023AH030020"]}]},{"name":"Open Research Fund of AnHui Key Laboratory of Detection Technology and Energy Saving Devices","award":["JCKJ2022A07"],"award-info":[{"award-number":["JCKJ2022A07"]}]},{"name":"Anhui Polytechnic University of Technology Introduced Talent Research Startup Fund","award":["2022YQQ009"],"award-info":[{"award-number":["2022YQQ009"]}]},{"name":"Youth Foundation of Anhui Polytechnic University","award":["Xjky2022039"],"award-info":[{"award-number":["Xjky2022039"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s00530-024-01318-8","type":"journal-article","created":{"date-parts":[[2024,4,8]],"date-time":"2024-04-08T15:02:12Z","timestamp":1712588532000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["CLDE-Net: crowd localization and density estimation based on CNN and transformer network"],"prefix":"10.1007","volume":"30","author":[{"given":"Yaocong","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanyuan","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huicheng","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bingyou","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guoyang","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinwen","family":"Hong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaobo","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,4,8]]},"reference":[{"key":"1318_CR1","doi-asserted-by":"crossref","unstructured":"Abousamra, S., Hoai, M., Samaras, D., Chen, C.: Localization in the crowd with topological constraints. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 872\u2013881 (2021)","DOI":"10.1609\/aaai.v35i2.16170"},{"key":"1318_CR2","doi-asserted-by":"crossref","unstructured":"Babu\u00a0Sam, D., Surya, S., Venkatesh\u00a0Babu, R.: Switching convolutional neural network for crowd counting. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5744\u20135752 (2017)","DOI":"10.1109\/CVPR.2017.429"},{"key":"1318_CR3","doi-asserted-by":"publisher","first-page":"71576","DOI":"10.1109\/ACCESS.2019.2918650","volume":"7","author":"S Basalamah","year":"2019","unstructured":"Basalamah, S., Khan, S.D., Ullah, H.: Scale driven convolutional neural network model for people counting and localization in crowd scenes. IEEE Access 7, 71576\u201371584 (2019). https:\/\/doi.org\/10.1109\/ACCESS.2019.2918650","journal-title":"IEEE Access"},{"key":"1318_CR4","doi-asserted-by":"crossref","unstructured":"Boominathan, L., Kruthiventi, S.S., Babu, R.V.: Crowdnet: a deep convolutional network for dense crowd counting. In: Proceedings of the 24th ACM International Conference on Multimedia, pp. 640\u2013644 (2016)","DOI":"10.1145\/2964284.2967300"},{"issue":"9","key":"1318_CR5","doi-asserted-by":"publisher","first-page":"4913","DOI":"10.1109\/TPAMI.2021.3076733","volume":"44","author":"J Cao","year":"2022","unstructured":"Cao, J., Pang, Y., Xie, J., Khan, F.S., Shao, L.: From handcrafted to deep features for pedestrian detection: a survey. IEEE Trans. Pattern Anal. Mach. Intell. 44(9), 4913\u20134934 (2022). https:\/\/doi.org\/10.1109\/TPAMI.2021.3076733","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"4","key":"1318_CR6","doi-asserted-by":"publisher","first-page":"2160","DOI":"10.1109\/TIP.2011.2172800","volume":"21","author":"AB Chan","year":"2012","unstructured":"Chan, A.B., Vasconcelos, N.: Counting people with low-level features and Bayesian regression. IEEE Trans. Image Process. 21(4), 2160\u20132177 (2012). https:\/\/doi.org\/10.1109\/TIP.2011.2172800","journal-title":"IEEE Trans. Image Process."},{"issue":"3","key":"1318_CR7","doi-asserted-by":"publisher","first-page":"1055","DOI":"10.1109\/TCSVT.2022.3208714","volume":"33","author":"Y Chen","year":"2023","unstructured":"Chen, Y., Yang, J., Chen, B., Du, S.: Counting varying density crowds through density guided adaptive selection CNN and transformer estimation. IEEE Trans. Circuits Syst. Video Technol. 33(3), 1055\u20131068 (2023). https:\/\/doi.org\/10.1109\/TCSVT.2022.3208714","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1318_CR8","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale (2020). arXiv preprint arXiv:2010.11929"},{"issue":"2","key":"1318_CR9","doi-asserted-by":"publisher","first-page":"865","DOI":"10.1109\/TKDE.2020.2985952","volume":"34","author":"Y Gong","year":"2022","unstructured":"Gong, Y., Li, Z., Zhang, J., Liu, W., Zheng, Y.: Online spatio-temporal crowd flow distribution prediction for complex metro system. IEEE Trans. Knowl. Data Eng. 34(2), 865\u2013880 (2022). https:\/\/doi.org\/10.1109\/TKDE.2020.2985952","journal-title":"IEEE Trans. Knowl. Data Eng."},{"issue":"12","key":"1318_CR10","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1109\/MCOM.2014.6979950","volume":"52","author":"F Hao","year":"2014","unstructured":"Hao, F., Jiao, M., Min, G., Yang, L.T.: A trajectory-based recruitment strategy of social sensors for participatory sensing. IEEE Commun. Mag. 52(12), 41\u201347 (2014). https:\/\/doi.org\/10.1109\/MCOM.2014.6979950","journal-title":"IEEE Commun. Mag."},{"key":"1318_CR11","doi-asserted-by":"publisher","first-page":"408","DOI":"10.1016\/j.future.2020.02.023","volume":"107","author":"F Hao","year":"2020","unstructured":"Hao, F., Pei, Z., Yang, L.T.: Diversified top-k maximal clique detection in social internet of things. Future Gener. Comput. Syst. 107, 408\u2013417 (2020). https:\/\/doi.org\/10.1016\/j.future.2020.02.023","journal-title":"Future Gener. Comput. Syst."},{"issue":"6","key":"1318_CR12","doi-asserted-by":"publisher","first-page":"3000","DOI":"10.1109\/TCSS.2023.3245075","volume":"10","author":"F Hao","year":"2023","unstructured":"Hao, F., Yang, Y., Shang, J., Park, D.S.: Afcminer: finding absolute fair cliques from attributed social networks for responsible computational social systems. IEEE Trans. Comput. Soc. Syst. 10(6), 3000\u20133011 (2023). https:\/\/doi.org\/10.1109\/TCSS.2023.3245075","journal-title":"IEEE Trans. Comput. Soc. Syst."},{"key":"1318_CR13","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"1318_CR14","doi-asserted-by":"publisher","first-page":"530","DOI":"10.1016\/j.jvcir.2016.03.021","volume":"38","author":"Y Hu","year":"2016","unstructured":"Hu, Y., Chang, H., Nian, F., Wang, Y., Li, T.: Dense crowd counting from still images with convolutional neural networks. J. Vis. Commun. Image Represent. 38, 530\u2013539 (2016)","journal-title":"J. Vis. Commun. Image Represent."},{"key":"1318_CR15","doi-asserted-by":"crossref","unstructured":"Idrees, H., Tayyab, M., Athrey, K., Zhang, D., Al-Maadeed, S., Rajpoot, N., Shah, M.: Composition loss for counting, density map estimation and localization in dense crowds. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 532\u2013546 (2018)","DOI":"10.1007\/978-3-030-01216-8_33"},{"issue":"9","key":"1318_CR16","doi-asserted-by":"publisher","first-page":"3119","DOI":"10.1109\/TCSVT.2019.2934989","volume":"30","author":"S Jiang","year":"2020","unstructured":"Jiang, S., Lu, X., Lei, Y., Liu, L.: Mask-aware networks for crowd counting. IEEE Trans. Circuits Syst. Video Technol. 30(9), 3119\u20133129 (2020). https:\/\/doi.org\/10.1109\/TCSVT.2019.2934989","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1318_CR17","doi-asserted-by":"crossref","unstructured":"Jiang, X., Xiao, Z., Zhang, B., Zhen, X., Cao, X., Doermann, D., Shao, L.: Crowd counting and density estimation by trellis encoder\u2013decoder networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6133\u20136142 (2019)","DOI":"10.1109\/CVPR.2019.00629"},{"key":"1318_CR18","doi-asserted-by":"publisher","first-page":"2127","DOI":"10.1007\/s00371-020-01974-7","volume":"37","author":"SD Khan","year":"2021","unstructured":"Khan, S.D., Basalamah, S.: Scale and density invariant head detection deep model for crowd counting in pedestrian crowds. Visual Comput. 37, 2127\u20132137 (2021)","journal-title":"Visual Comput."},{"key":"1318_CR19","doi-asserted-by":"publisher","first-page":"3051","DOI":"10.1007\/s13369-020-04990-w","volume":"46","author":"SD Khan","year":"2021","unstructured":"Khan, S.D., Basalamah, S.: Sparse to dense scale prediction for crowd counting in high density crowds. Arab. J. Sci. Eng. 46, 3051\u20133065 (2021)","journal-title":"Arab. J. Sci. Eng."},{"key":"1318_CR20","doi-asserted-by":"publisher","first-page":"168","DOI":"10.1007\/s44196-021-00016-x","volume":"14","author":"SD Khan","year":"2021","unstructured":"Khan, S.D., Salih, Y., Zafar, B., Noorwali, A.: A deep-fusion network for crowd counting in high-density crowded scenes. Int. J. Comput. Intell. Syst. 14, 168 (2021)","journal-title":"Int. J. Comput. Intell. Syst."},{"issue":"1\u20132","key":"1318_CR21","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1002\/nav.3800020109","volume":"2","author":"HW Kuhn","year":"1955","unstructured":"Kuhn, H.W.: The Hungarian method for the assignment problem. Naval Res. Logist. Q. 2(1\u20132), 83\u201397 (1955)","journal-title":"Naval Res. Logist. Q."},{"key":"1318_CR22","doi-asserted-by":"crossref","unstructured":"Laradji, I.H., Rostamzadeh, N., Pinheiro, P.O., Vazquez, D., Schmidt, M.: Where are the blobs: counting by localization with point supervision. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 547\u2013562 (2018)","DOI":"10.1007\/978-3-030-01216-8_34"},{"key":"1318_CR23","unstructured":"Lempitsky, V., Zisserman, A.: Learning to count objects in images. In: Advances in Neural Information Processing Systems, vol. 23 (2010)"},{"key":"1318_CR24","doi-asserted-by":"publisher","first-page":"2671","DOI":"10.1007\/s00371-022-02485-3","volume":"39","author":"B Li","year":"2022","unstructured":"Li, B., Zhang, Y., Xu, H., Yin, B.: Ccst: crowd counting with swin transformer. Visual Comput. 39, 2671\u20132682 (2022)","journal-title":"Visual Comput."},{"key":"1318_CR25","doi-asserted-by":"publisher","first-page":"275","DOI":"10.1109\/TIP.2021.3130545","volume":"31","author":"J Li","year":"2022","unstructured":"Li, J., Huang, Q., Du, Y., Zhen, X., Chen, S., Shao, L.: Variational abnormal behavior detection with motion consistency. IEEE Trans. Image Process. 31, 275\u2013286 (2022). https:\/\/doi.org\/10.1109\/TIP.2021.3130545","journal-title":"IEEE Trans. Image Process."},{"key":"1318_CR26","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhang, X., Chen, D.: Csrnet: dilated convolutional neural networks for understanding the highly congested scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00120"},{"issue":"6","key":"1318_CR27","doi-asserted-by":"publisher","first-page":"160104","DOI":"10.1007\/s11432-021-3445-y","volume":"65","author":"D Liang","year":"2022","unstructured":"Liang, D., Chen, X., Xu, W., Zhou, Y., Bai, X.: Transcrowd: weakly-supervised crowd counting with transformers. Sci. China Inf. Sci. 65(6), 160104 (2022)","journal-title":"Sci. China Inf. Sci."},{"key":"1318_CR28","doi-asserted-by":"publisher","first-page":"6040","DOI":"10.1109\/TMM.2022.3203870","volume":"25","author":"D Liang","year":"2022","unstructured":"Liang, D., Xu, W., Zhu, Y., Zhou, Y.: Focal inverse distance transform maps for crowd localization. IEEE Trans. Multimed. 25, 6040\u20136052 (2022)","journal-title":"IEEE Trans. Multimed."},{"key":"1318_CR29","doi-asserted-by":"crossref","unstructured":"Lin, H., Ma, Z., Ji, R., Wang, Y., Hong, X.: Boosting crowd counting via multifaceted attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 19628\u201319637 (2022)","DOI":"10.1109\/CVPR52688.2022.01901"},{"issue":"6","key":"1318_CR30","doi-asserted-by":"publisher","first-page":"645","DOI":"10.1109\/3468.983420","volume":"31","author":"SF Lin","year":"2001","unstructured":"Lin, S.F., Chen, J.Y., Chao, H.X.: Estimation of number of people in crowded scenes using perspective transformation. IEEE Trans. Syst. Man Cybern. Part A Syst. Hum. 31(6), 645\u2013654 (2001)","journal-title":"IEEE Trans. Syst. Man Cybern. Part A Syst. Hum."},{"key":"1318_CR31","doi-asserted-by":"crossref","unstructured":"Liu, C., Weng, X., Mu, Y.: Recurrent attentive zooming for joint crowd counting and precise localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00131"},{"key":"1318_CR32","doi-asserted-by":"publisher","unstructured":"Liu, J., Gao, C., Meng, D., Hauptmann, A.G.: Decidenet: counting varying density crowds through attention guided detection and density estimation. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5197\u20135206 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00545","DOI":"10.1109\/CVPR.2018.00545"},{"key":"1318_CR33","doi-asserted-by":"crossref","unstructured":"Liu, L., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1318_CR34","doi-asserted-by":"crossref","unstructured":"Liu, W., Salzmann, M., Fua, P.: Context-aware crowd counting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5099\u20135108 (2019)","DOI":"10.1109\/CVPR.2019.00524"},{"key":"1318_CR35","unstructured":"Liu, X., Li, G., Qi, Y., Han, Z., Huang, Q., Yang, M.H., Sebe, N.: Consistency-aware anchor pyramid network for crowd localization (2022). arXiv preprint arXiv:2212.04067"},{"key":"1318_CR36","doi-asserted-by":"publisher","unstructured":"Liu, Y., Shi, M., Zhao, Q., Wang, X.: Point in, box out: beyond counting persons in crowds. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6462\u20136471 (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00663","DOI":"10.1109\/CVPR.2019.00663"},{"issue":"3","key":"1318_CR37","doi-asserted-by":"publisher","first-page":"3422","DOI":"10.1109\/TITS.2022.3226504","volume":"24","author":"Y Meng","year":"2023","unstructured":"Meng, Y., Bridge, J., Zhao, Y., Joddrell, M., Qiao, Y., Yang, X., Huang, X., Zheng, Y.: Transportation object counting with graph-based adaptive auxiliary learning. IEEE Trans. Intell. Transp. Syst. 24(3), 3422\u20133437 (2023). https:\/\/doi.org\/10.1109\/TITS.2022.3226504","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"1318_CR38","doi-asserted-by":"publisher","unstructured":"Ranjan, A., Pathare, N., Dhavale, S., Kumar, S.: Performance analysis of yolo algorithms for real-time crowd counting. In: 2022 2nd Asian Conference on Innovation in Technology (ASIANCON), pp. 1\u20138 (2022). https:\/\/doi.org\/10.1109\/ASIANCON55314.2022.9909018","DOI":"10.1109\/ASIANCON55314.2022.9909018"},{"key":"1318_CR39","doi-asserted-by":"publisher","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 779\u2013788 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.91","DOI":"10.1109\/CVPR.2016.91"},{"key":"1318_CR40","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"issue":"8","key":"1318_CR41","doi-asserted-by":"publisher","first-page":"2739","DOI":"10.1109\/TPAMI.2020.2974830","volume":"43","author":"DB Sam","year":"2021","unstructured":"Sam, D.B., Peri, S.V., Sundararaman, M.N., Kamath, A., Babu, R.V.: Locate, size, and count: accurately resolving people in dense crowds via detection. IEEE Trans. Pattern Anal. Mach. Intell. 43(8), 2739\u20132751 (2021). https:\/\/doi.org\/10.1109\/TPAMI.2020.2974830","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1318_CR42","doi-asserted-by":"publisher","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.C.: Mobilenetv2: inverted residuals and linear bottlenecks. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00474","DOI":"10.1109\/CVPR.2018.00474"},{"key":"1318_CR43","doi-asserted-by":"crossref","unstructured":"Shi, M., Yang, Z., Xu, C., Chen, Q.: Revisiting perspective information for efficient crowd counting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7279\u20137288 (2019)","DOI":"10.1109\/CVPR.2019.00745"},{"key":"1318_CR44","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition (2014). arXiv preprint arXiv:1409.1556"},{"key":"1318_CR45","doi-asserted-by":"crossref","unstructured":"Song, Q., Wang, C., Jiang, Z., Wang, Y., Tai, Y., Wang, C., Li, J., Huang, F., Wu, Y.: Rethinking counting and localization in crowds: a purely point-based framework. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3365\u20133374 (2021)","DOI":"10.1109\/ICCV48922.2021.00335"},{"key":"1318_CR46","doi-asserted-by":"crossref","unstructured":"Song, Q., Wang, C., Wang, Y., Tai, Y., Wang, C., Li, J., Wu, J., Ma, J.: To choose or to fuse? Scale selection for crowd counting. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 2576\u20132583 (2021)","DOI":"10.1609\/aaai.v35i3.16360"},{"key":"1318_CR47","doi-asserted-by":"publisher","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2818\u20132826 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.308","DOI":"10.1109\/CVPR.2016.308"},{"key":"1318_CR48","unstructured":"Tian, Y., Chu, X., Wang, H.: Cctrans: simplifying and improving crowd counting with transformer (2021). arXiv preprint arXiv:2109.14483"},{"key":"1318_CR49","doi-asserted-by":"publisher","first-page":"2114","DOI":"10.1109\/TIP.2021.3049938","volume":"30","author":"J Wan","year":"2021","unstructured":"Wan, J., Kumar, N.S., Chan, A.B.: Fine-grained crowd counting. IEEE Trans. Image Process. 30, 2114\u20132126 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"1318_CR50","doi-asserted-by":"publisher","unstructured":"Wan, J., Liu, Z., Chan, A.B.: A generalized loss function for crowd counting and localization. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1974\u20131983 (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.00201","DOI":"10.1109\/CVPR46437.2021.00201"},{"key":"1318_CR51","unstructured":"Wang, F., Liu, K., Long, F., Sang, N., Xia, X., Sang, J.: Joint CNN and transformer network via weakly supervised learning for efficient crowd counting (2022). arXiv preprint arXiv:2203.06388"},{"key":"1318_CR52","doi-asserted-by":"publisher","first-page":"292","DOI":"10.1016\/j.neucom.2020.05.056","volume":"407","author":"P Wang","year":"2020","unstructured":"Wang, P., Gao, C., Wang, Y., Li, H., Gao, Y.: Mobilecount: an efficient encoder-decoder framework for real-time crowd counting. Neurocomputing 407, 292\u2013299 (2020)","journal-title":"Neurocomputing"},{"issue":"6","key":"1318_CR53","doi-asserted-by":"publisher","first-page":"2141","DOI":"10.1109\/TPAMI.2020.3013269","volume":"43","author":"Q Wang","year":"2020","unstructured":"Wang, Q., Gao, J., Lin, W., Li, X.: NWPU-crowd: a large-scale benchmark for crowd counting and localization. IEEE Trans. Pattern Anal. Mach. Intell. 43(6), 2141\u20132149 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1318_CR54","doi-asserted-by":"publisher","unstructured":"Wang, S., Chang, J., Li, H., Wang, Z., Ouyang, W., Tian, Q.: Open-set fine-grained retrieval via prompting vision-language evaluator. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 19381\u201319391 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01857","DOI":"10.1109\/CVPR52729.2023.01857"},{"issue":"5","key":"1318_CR55","doi-asserted-by":"publisher","first-page":"4695","DOI":"10.1109\/TITS.2021.3055207","volume":"23","author":"S Wang","year":"2022","unstructured":"Wang, S., Miao, H., Li, J., Cao, J.: Spatio-temporal knowledge transfer for urban crowd flow prediction via deep attentive adaptation networks. IEEE Trans. Intell. Transp. Syst. 23(5), 4695\u20134705 (2022). https:\/\/doi.org\/10.1109\/TITS.2021.3055207","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"1318_CR56","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/s11263-023-01873-z","volume":"132","author":"S Wang","year":"2023","unstructured":"Wang, S., Wang, Z., Li, H., Chang, J., Ouyang, W., Tian, Q.: Accurate fine-grained object recognition with structure-driven relation graph networks. Int. J. Comput. Vis. 132, 137\u2013160 (2023)","journal-title":"Int. J. Comput. Vis."},{"key":"1318_CR57","doi-asserted-by":"publisher","first-page":"2876","DOI":"10.1109\/TIP.2021.3055632","volume":"30","author":"Y Wang","year":"2021","unstructured":"Wang, Y., Hou, J., Hou, X., Chau, L.P.: A self-training approach for point-supervised object detection and counting in crowds. IEEE Trans. Image Process. 30, 2876\u20132887 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"1318_CR58","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3193961","author":"Z Wang","year":"2022","unstructured":"Wang, Z., Li, Z., Leng, J., Li, M., Bai, L.: Multiple pedestrian tracking with graph attention map on urban road scene. IEEE Trans. Intell. Transp. Syst. (2022). https:\/\/doi.org\/10.1109\/TITS.2022.3193961","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"1318_CR59","doi-asserted-by":"crossref","unstructured":"Wang, Z., Wang, S., Li, H., Dou, Z., Li, J.: Graph-propagation based correlation learning for weakly supervised fine-grained image classification. In: AAAI Conference on Artificial Intelligence (2020). https:\/\/api.semanticscholar.org\/CorpusID:214471460","DOI":"10.1609\/aaai.v34i07.6912"},{"key":"1318_CR60","doi-asserted-by":"publisher","unstructured":"Wang, Z., Wang, S., Yang, S., Li, H., Li, J., Li, Z.: Weakly supervised fine-grained image classification via Gaussian mixture model oriented discriminative learning. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9746\u20139755 (2020). https:\/\/doi.org\/10.1109\/CVPR42600.2020.00977","DOI":"10.1109\/CVPR42600.2020.00977"},{"key":"1318_CR61","doi-asserted-by":"crossref","unstructured":"Wu, B., Nevatia, R.: Detection of multiple, partially occluded humans in a single image by Bayesian combination of edgelet part detectors. In: Tenth IEEE International Conference on Computer Vision (ICCV\u201905), vol.\u00a01, pp. 90\u201397. IEEE (2005)","DOI":"10.1109\/ICCV.2005.74"},{"issue":"11","key":"1318_CR62","doi-asserted-by":"publisher","first-page":"21548","DOI":"10.1109\/TITS.2022.3186707","volume":"23","author":"Y Xie","year":"2022","unstructured":"Xie, Y., Niu, J., Zhang, Y., Ren, F.: Multisize patched spatial-temporal transformer network for short- and long-term crowd flow prediction. IEEE Trans. Intell. Transp. Syst. 23(11), 21548\u201321568 (2022). https:\/\/doi.org\/10.1109\/TITS.2022.3186707","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"10","key":"1318_CR63","doi-asserted-by":"publisher","first-page":"2814","DOI":"10.1109\/TCSVT.2017.2731866","volume":"28","author":"M Xu","year":"2018","unstructured":"Xu, M., Li, C., Lv, P., Lin, N., Hou, R., Zhou, B.: An efficient method of crowd aggregation computation in public areas. IEEE Trans. Circuits Syst. Video Technol. 28(10), 2814\u20132825 (2018). https:\/\/doi.org\/10.1109\/TCSVT.2017.2731866","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1318_CR64","doi-asserted-by":"crossref","unstructured":"Yang, S., Guo, W., Ren, Y.: Crowdformer: an overlap patching vision transformer for top-down crowd counting. In: Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence (IJCAI-22), pp. 1545\u20131551 (2022)","DOI":"10.24963\/ijcai.2022\/215"},{"issue":"3","key":"1318_CR65","doi-asserted-by":"publisher","first-page":"1020","DOI":"10.1109\/TNSE.2021.3067939","volume":"9","author":"Y Yang","year":"2022","unstructured":"Yang, Y., Hao, F., Pang, B., Min, G., Wu, Y.: Dynamic maximal cliques detection and evolution management in social internet of things: a formal concept analysis approach. IEEE Trans. Netw. Sci. Eng. 9(3), 1020\u20131032 (2022). https:\/\/doi.org\/10.1109\/TNSE.2021.3067939","journal-title":"IEEE Trans. Netw. Sci. Eng."},{"key":"1318_CR66","unstructured":"Zhang, C., Li, H., Wang, X., Yang, X.: Cross-scene crowd counting via deep convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 833\u2013841 (2015)"},{"key":"1318_CR67","doi-asserted-by":"crossref","unstructured":"Zhang, L., Shi, M., Chen, Q.: Crowd counting via scale-adaptive convolutional neural network. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1113\u20131121. IEEE (2018)","DOI":"10.1109\/WACV.2018.00127"},{"key":"1318_CR68","doi-asserted-by":"crossref","unstructured":"Zhang, S., Wen, L., Bian, X., Lei, Z., Li, S.Z.: Occlusion-aware r-cnn: detecting pedestrians in a crowd. In: Proceedings of the European Conference on Computer Vision (ECCV) (2018)","DOI":"10.1007\/978-3-030-01219-9_39"},{"key":"1318_CR69","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3221622","author":"X Zhang","year":"2022","unstructured":"Zhang, X., Fang, J., Yang, B., Chen, S., Li, B.: Hybrid attention and motion constraint for anomaly detection in crowded scenes. IEEE Trans. Circuits Syst. Video Technol. (2022). https:\/\/doi.org\/10.1109\/TCSVT.2022.3221622","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1318_CR70","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Zhou, D., Chen, S., Gao, S., Ma, Y.: Single-image crowd counting via multi-column convolutional neural network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 589\u2013597 (2016)","DOI":"10.1109\/CVPR.2016.70"},{"issue":"5","key":"1318_CR71","doi-asserted-by":"publisher","first-page":"1183","DOI":"10.1109\/TMM.2018.2875360","volume":"21","author":"Q Zhou","year":"2019","unstructured":"Zhou, Q., Zhong, B., Zhang, Y., Li, J., Fu, Y.: Deep alignment network based multi-person tracking with occlusion and motion reasoning. IEEE Trans. Multimed. 21(5), 1183\u20131194 (2019). https:\/\/doi.org\/10.1109\/TMM.2018.2875360","journal-title":"IEEE Trans. Multimed."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01318-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01318-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01318-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,15]],"date-time":"2024-11-15T22:05:48Z","timestamp":1731708348000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01318-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,8]]},"references-count":71,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["1318"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01318-8","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4,8]]},"assertion":[{"value":"1 December 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 April 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"120"}}