{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T20:34:56Z","timestamp":1776890096096,"version":"3.51.2"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2022,9,29]],"date-time":"2022-09-29T00:00:00Z","timestamp":1664409600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,9,29]],"date-time":"2022-09-29T00:00:00Z","timestamp":1664409600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"Research Grants Council of the Hong Kong Special Administrative Region, China","award":["T32-101\/15-R"],"award-info":[{"award-number":["T32-101\/15-R"]}]},{"name":"Research Grants Council of the Hong Kong Special Administrative Region, China","award":["CityU 11212518"],"award-info":[{"award-number":["CityU 11212518"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s11263-022-01685-7","type":"journal-article","created":{"date-parts":[[2022,9,29]],"date-time":"2022-09-29T10:06:01Z","timestamp":1664445961000},"page":"3123-3139","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["3D Crowd Counting via Geometric Attention-Guided Multi-view Fusion"],"prefix":"10.1007","volume":"130","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6212-9799","authenticated-orcid":false,"given":"Qi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2886-2513","authenticated-orcid":false,"given":"Antoni B.","family":"Chan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,29]]},"reference":[{"key":"1685_CR1","doi-asserted-by":"crossref","unstructured":"Bai, S., He, Z., Qiao, Y., Hu, H., Wu, W., & Yan, J. (2020). Adaptive dilated network with self-correction supervision for counting. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 4594\u20134603).","DOI":"10.1109\/CVPR42600.2020.00465"},{"key":"1685_CR2","doi-asserted-by":"crossref","unstructured":"Boominathan, L., Kruthiventi, S. S., & Babu, R. V. (2016). Crowdnet: A deep convolutional network for dense crowd counting. In ACM multimedia conference. ACM (pp. 640\u2013644).","DOI":"10.1145\/2964284.2967300"},{"key":"1685_CR3","doi-asserted-by":"crossref","unstructured":"Cao, X., & Wang, Z., et\u00a0al. (2018). Scale aggregation network for accurate and efficient crowd counting. In ECCV (pp. 734\u2013750).","DOI":"10.1007\/978-3-030-01228-1_45"},{"key":"1685_CR4","doi-asserted-by":"crossref","unstructured":"Chan, A. B., Liang, Z. S. J., & Vasconcelos, N. (2008). Privacy preserving crowd monitoring: Counting people without people models or tracking. In CVPR (pp. 1\u20137).","DOI":"10.1109\/CVPR.2008.4587569"},{"key":"1685_CR5","unstructured":"Chang, A. X., et\u00a0al. (2015). Shapenet: An information-rich 3d model repository. arXiv preprint arXiv:1512.03012."},{"key":"1685_CR6","doi-asserted-by":"crossref","unstructured":"Chen, K., Chen, L. C., Gong, S., & Xiang, T. (2012). Feature mining for localised crowd counting. In BMVC.","DOI":"10.5244\/C.26.21"},{"key":"1685_CR7","doi-asserted-by":"crossref","unstructured":"Choy, C. B., Xu, D., Gwak, J., Chen, K., & Savarese, S. (2016). 3d-r2n2: A unified approach for single and multi-view 3d object reconstruction. In ECCV. Springer (pp. 628\u2013644).","DOI":"10.1007\/978-3-319-46484-8_38"},{"key":"1685_CR8","unstructured":"Dittrich, F., de\u00a0Oliveira, L. E., Britto,\u00a0Jr A. S., & Koerich, A. L. (2017). People counting in crowded and outdoor scenes using a hybrid multi-camera approach. arXiv preprint arXiv:1704.00326."},{"key":"1685_CR9","doi-asserted-by":"crossref","unstructured":"Ferryman, J., & Shahrokni, A. (2009). Pets2009: Dataset and challenge. In IEEE international workshop on performance evaluation of tracking and surveillance (pp. 1\u20136).","DOI":"10.1109\/PETS-WINTER.2009.5399556"},{"key":"1685_CR10","doi-asserted-by":"crossref","unstructured":"Ge, W., & Collins, R. T. (2010). Crowd detection with a multiview sampler. In ECCV (pp. 324\u2013337).","DOI":"10.1007\/978-3-642-15555-0_24"},{"key":"1685_CR11","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Fouhey, D. F., Rodriguez, M., & Gupta, A. (2016). Learning a predictable and generative vector representation for objects. In ECCV. Springer (pp. 484\u2013499).","DOI":"10.1007\/978-3-319-46466-4_29"},{"key":"1685_CR12","doi-asserted-by":"crossref","unstructured":"Huang, P. H., & Matzen, K., et\u00a0al. (2018). Deepmvs: Learning multi-view stereopsis. In CVPR (pp. 2821\u20132830).","DOI":"10.1109\/CVPR.2018.00298"},{"key":"1685_CR13","doi-asserted-by":"crossref","unstructured":"Idrees, H., et\u00a0al. (2018). Composition loss for counting, density map estimation and localization in dense crowds. In ECCV (pp. 532\u2013546).","DOI":"10.1007\/978-3-030-01216-8_33"},{"key":"1685_CR14","doi-asserted-by":"crossref","unstructured":"Idrees, H., Saleemi, I., Seibert, C., & Shah, M. (2013). Multi-source multi-scale counting in extremely dense crowd images. In CVPR (pp. 2547\u20132554).","DOI":"10.1109\/CVPR.2013.329"},{"key":"1685_CR15","doi-asserted-by":"crossref","unstructured":"Iskakov, K., Burkov, E., Lempitsky, V., & Malkov, Y. (2019). Learnable triangulation of human pose. In ICCV.","DOI":"10.1109\/ICCV.2019.00781"},{"key":"1685_CR16","unstructured":"Jaderberg, M., Simonyan, K., Zisserman, A., & Kavukcuoglu, K. (2015). Spatial transformer networks. In Advances in neural information processing systems (pp. 2017\u20132025)."},{"key":"1685_CR17","doi-asserted-by":"crossref","unstructured":"Jiang, X., et\u00a0al. (2019). Crowd counting and density estimation by trellis encoder-decoder networks. In CVPR (pp. 6133\u20136142).","DOI":"10.1109\/CVPR.2019.00629"},{"key":"1685_CR18","doi-asserted-by":"crossref","unstructured":"Jiang, X., Zhang, L., Xu, M., Zhang, T., Lv, P., Zhou, B., Yang, X., & Pang, Y. (2020). Attention scaling for crowd counting. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 4706\u20134715).","DOI":"10.1109\/CVPR42600.2020.00476"},{"key":"1685_CR19","unstructured":"Kang, D., & Chan, A. (2018). Crowd counting by adaptively fusing predictions from an image pyramid. In BMVC."},{"key":"1685_CR20","unstructured":"Kang, D., Dhar, D., & Chan, A. (2017). Incorporating side information by adaptive convolution. In Advances in neural information processing systems (pp. 3867\u20133877)."},{"key":"1685_CR21","unstructured":"Kar, A., H\u00e4ne, C., & Malik, J. (2017). Learning a multi-view stereo machine. In NIPS (pp. 365\u2013376)."},{"key":"1685_CR22","doi-asserted-by":"crossref","unstructured":"Li, J., Huang, L., & Liu, C. (2012). People counting across multiple cameras for intelligent video surveillance. In IEEE ninth international conference on advanced video and signal-based surveillance (AVSS). IEEE (pp. 178\u2013183).","DOI":"10.1109\/AVSS.2012.54"},{"key":"1685_CR23","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhang, X., & Chen, D. (2018). Csrnet: Dilated convolutional neural networks for understanding the highly congested scenes. In CVPR (pp. 1091\u20131100).","DOI":"10.1109\/CVPR.2018.00120"},{"key":"1685_CR24","doi-asserted-by":"crossref","unstructured":"Lian, D., Li, J., Zheng, J., Luo, W., & Gao, S. (2019). Density map regression guided detection network for rgb-d crowd counting and localization. In CVPR (pp. 1821\u20131830).","DOI":"10.1109\/CVPR.2019.00192"},{"key":"1685_CR25","doi-asserted-by":"crossref","unstructured":"Liao, S., Hu, Y., Zhu, X., & Li, S. Z. (2015). Person re-identification by local maximal occurrence representation and metric learning. In CVPR (pp. 2197\u20132206).","DOI":"10.1109\/CVPR.2015.7298832"},{"key":"1685_CR26","doi-asserted-by":"crossref","unstructured":"Lim, J. J., Pirsiavash, H., & Torralba, A. (2013). Parsing IKEA objects: Fine pose estimation. In ICCV.","DOI":"10.1109\/ICCV.2013.372"},{"key":"1685_CR27","doi-asserted-by":"crossref","unstructured":"Liu, C., et\u00a0al. (2019). Recurrent attentive zooming for joint crowd counting and precise localization. In CVPR (pp. 1217\u20131226).","DOI":"10.1109\/CVPR.2019.00131"},{"key":"1685_CR28","doi-asserted-by":"crossref","unstructured":"Liu, J., Gao, C., Meng, D., Hauptmann, A. G. (2018). Decidenet: Counting varying density crowds through attention guided detection and density estimation. In CVPR (pp. 5197\u20135206).","DOI":"10.1109\/CVPR.2018.00545"},{"key":"1685_CR29","doi-asserted-by":"crossref","unstructured":"Liu, W., Salzmann, M., Fua, P. (2019). Context-aware crowd counting. In CVPR (pp. 5099\u20135108).","DOI":"10.1109\/CVPR.2019.00524"},{"key":"1685_CR30","doi-asserted-by":"crossref","unstructured":"Liu, X., Yang, J., Ding, W. (2020). Adaptive mixture regression network with local counting map for crowd counting. arXiv preprint arXiv:2005.05776.","DOI":"10.1007\/978-3-030-58586-0_15"},{"issue":"2","key":"1685_CR31","first-page":"31","volume":"3","author":"H Ma","year":"2012","unstructured":"Ma, H., Zeng, C., & Ling, C. X. (2012). A reliable people counting system via multiple cameras. ACM Transactions on Intelligent Systems and Technology (TIST), 3(2), 31.","journal-title":"ACM Transactions on Intelligent Systems and Technology (TIST)"},{"key":"1685_CR32","doi-asserted-by":"crossref","unstructured":"Ma, Z., Wei, X., Hong, X., & Gong, Y. (2019). Bayesian loss for crowd count estimation with point supervision. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 6142\u20136151).","DOI":"10.1109\/ICCV.2019.00624"},{"key":"1685_CR33","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1016\/j.patrec.2013.10.006","volume":"36","author":"L Maddalena","year":"2014","unstructured":"Maddalena, L., Petrosino, A., & Russo, F. (2014). People counting by learning their appearance in a multi-view camera environment. Pattern Recognition Letters, 36, 125\u2013134.","journal-title":"Pattern Recognition Letters"},{"key":"1685_CR34","doi-asserted-by":"crossref","unstructured":"Onoro-Rubio, D., L\u00f3pez-Sastre, R. J. (2016). Towards perspective-free object counting with deep learning. In ECCV. Springer (pp .615\u2013629).","DOI":"10.1007\/978-3-319-46478-7_38"},{"key":"1685_CR35","doi-asserted-by":"crossref","unstructured":"Ranjan, V., Le, H., & Hoai, M. (2018). Iterative crowd counting. In ECCV (pp. 270\u2013285).","DOI":"10.1007\/978-3-030-01234-2_17"},{"key":"1685_CR36","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2015). Faster R-CNN: Towards real-time object detection with region proposal networks. In Advances in neural information processing systems (pp. 91\u201399)."},{"key":"1685_CR37","doi-asserted-by":"crossref","unstructured":"Ristani, E., & Solera, F., et\u00a0al. (2016). Performance measures and a data set for multi-target, multi-camera tracking. In ECCV workshop on benchmarking multi-target tracking.","DOI":"10.1007\/978-3-319-48881-3_2"},{"issue":"8","key":"1685_CR38","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1016\/j.patrec.2013.10.002","volume":"44","author":"D Ryan","year":"2014","unstructured":"Ryan, D., Denman, S., Fookes, C., & Sridharan, S. (2014). Scene invariant multi camera crowd counting. Pattern Recognition Letters, 44(8), 98\u2013112.","journal-title":"Pattern Recognition Letters"},{"key":"1685_CR39","doi-asserted-by":"crossref","unstructured":"Sam, D. B., Surya, S., & Babu, R. V. (2017). Switching convolutional neural network for crowd counting. In CVPR (pp. 4031\u20134039).","DOI":"10.1109\/CVPR.2017.429"},{"key":"1685_CR40","doi-asserted-by":"crossref","unstructured":"Shen, Z., Xu, Y., Ni, B., Wang, M., Hu, J., & Yang, X. (2018). Crowd counting via adversarial cross-scale consistency pursuit. In CVPR (pp. 5245\u20135254).","DOI":"10.1109\/CVPR.2018.00550"},{"key":"1685_CR41","doi-asserted-by":"crossref","unstructured":"Shi, M., & Yang, Z., et\u00a0al. (2019). Revisiting perspective information for efficient crowd counting. In CVPR (pp. 7279\u20137288).","DOI":"10.1109\/CVPR.2019.00745"},{"key":"1685_CR42","doi-asserted-by":"crossref","unstructured":"Sindagi, V. A., & Patel, V. M. (2017). Generating high-quality crowd density maps using contextual pyramid cnns. In ICCV (pp. 1879\u20131888).","DOI":"10.1109\/ICCV.2017.206"},{"key":"1685_CR43","doi-asserted-by":"crossref","unstructured":"Sindagi, V. A., Yasarla, R., Babu, D. S., Babu, R. V., & Patel, V. M. (2020). Learning to count in the crowd from limited labeled data. arXiv preprint arXiv:2007.03195.","DOI":"10.1007\/978-3-030-58621-8_13"},{"key":"1685_CR44","doi-asserted-by":"crossref","unstructured":"Sitzmann, V., Thies, J., Heide, F., Nie\u00dfner, M., Wetzstein, G., & Zollh\u00f6fer, M. (2019). Deepvoxels: Learning persistent 3d feature embeddings. In Proceedings of computer vision and pattern recognition (CVPR). IEEE.","DOI":"10.1109\/CVPR.2019.00254"},{"issue":"1","key":"1685_CR45","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1109\/TIP.2014.2363445","volume":"24","author":"N Tang","year":"2014","unstructured":"Tang, N., Lin, Y. Y., Weng, M. F., & Liao, H. Y. (2014). Cross-camera knowledge transfer for multiview people counting. IEEE Transactions on Image Processing, 24(1), 80\u201393.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1685_CR46","unstructured":"Wang, B., Liu, H., Samaras, D., & Hoai, M. (2020). Distribution matching for crowd counting. arXiv preprint arXiv:2009.13077."},{"key":"1685_CR47","doi-asserted-by":"crossref","unstructured":"Wang, Q., & Gao, J., et\u00a0al. (2019). Learning from synthetic data for crowd counting in the wild. In CVPR (pp. 8198\u20138207).","DOI":"10.1109\/CVPR.2019.00839"},{"key":"1685_CR48","doi-asserted-by":"crossref","unstructured":"Xiong, H., Lu, H., Liu, C., Liu, L., Cao, Z., & Shen, C. (2019). From open set to closed set: Counting objects by spatial divide-and-conquer. In Proceedings of the IEEE\/CVF international conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2019.00845"},{"key":"1685_CR49","unstructured":"Yan, X., & Yang, J., et\u00a0al. (2016). Perspective transformer nets: Learning single-view 3d object reconstruction without 3d supervision. In NIPS (pp. 1696\u20131704)."},{"key":"1685_CR50","doi-asserted-by":"crossref","unstructured":"Yang, Y., Li, G., Wu, Z., Su, L., Huang, Q., & Sebe, N. (2020). Reverse perspective network for perspective-aware object counting. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 4374\u20134383).","DOI":"10.1109\/CVPR42600.2020.00443"},{"key":"1685_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, C., & Li. H., et\u00a0al. (2015). Cross-scene crowd counting via deep convolutional neural networks. In CVPR (pp. 833\u2013841).","DOI":"10.1109\/CVPR.2015.7298684"},{"key":"1685_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, Q., & Chan, A. B. (2019). Wide-area crowd counting via ground-plane density maps and multi-view fusion cnns. In CVPR (pp. 8297\u20138306).","DOI":"10.1109\/CVPR.2019.00849"},{"key":"1685_CR53","doi-asserted-by":"crossref","unstructured":"Zhang, Q., & Chan, A. B. (2020). 3d crowd counting via multi-view fusion with 3d gaussian kernels. In AAAI (pp. 12837\u201312844).","DOI":"10.1609\/aaai.v34i07.6980"},{"key":"1685_CR54","doi-asserted-by":"crossref","unstructured":"Zhang, Q., & Chan, A. B. (2021). Cross-view cross-scene multi-view crowd counting. In Submitted to CVPR 2021.","DOI":"10.1109\/CVPR46437.2021.00062"},{"key":"1685_CR55","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et\u00a0al. (2016). Single-image crowd counting via multi-column convolutional neural network. In CVPR (pp. 589\u2013597).","DOI":"10.1109\/CVPR.2016.70"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-022-01685-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-022-01685-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-022-01685-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,26]],"date-time":"2022-10-26T08:19:21Z","timestamp":1666772361000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-022-01685-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,9,29]]},"references-count":55,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["1685"],"URL":"https:\/\/doi.org\/10.1007\/s11263-022-01685-7","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,9,29]]},"assertion":[{"value":"19 February 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 September 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 September 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}