{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T14:40:38Z","timestamp":1773931238232,"version":"3.50.1"},"reference-count":65,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,10]]},"DOI":"10.1007\/s11263-021-01499-z","type":"journal-article","created":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T06:02:50Z","timestamp":1626588170000},"page":"2731-2744","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":37,"title":["Cascaded Split-and-Aggregate Learning with Feature Recombination for Pedestrian Attribute Recognition"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0559-5464","authenticated-orcid":false,"given":"Yang","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zichang","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2851-4260","authenticated-orcid":false,"given":"Prayag","family":"Tiwari","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hari Mohan","family":"Pandey","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Wan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Lei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guodong","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stan Z.","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,7,18]]},"reference":[{"key":"1499_CR1","doi-asserted-by":"crossref","unstructured":"Chen, B., Deng, W., & Hu, J. (2019). Mixed high-order attention network for person re-identification. In ICCV.","DOI":"10.1109\/ICCV.2019.00046"},{"key":"1499_CR2","doi-asserted-by":"crossref","unstructured":"Chen, L. C., Papandreou, G., Kokkinos, I., Murphy, K., & Yuille, A. L. (2017). Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE TPAMI.","DOI":"10.1109\/TPAMI.2017.2699184"},{"key":"1499_CR3","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., & Fei-Fei, L. (2009) . Imagenet: A large-scale hierarchical image database. In CVPR.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1499_CR4","doi-asserted-by":"crossref","unstructured":"Deng, Y., Luo, P., Loy, C. C., & Tang, X. (2014). Pedestrian attribute recognition at far distance. In ACM MM.","DOI":"10.1145\/2647868.2654966"},{"key":"1499_CR5","doi-asserted-by":"crossref","unstructured":"Dixit, M., Kwitt, R., Niethammer, M., & Vasconcelos, N. (2017). Aga: Attribute-guided augmentation. In CVPR.","DOI":"10.1109\/CVPR.2017.355"},{"key":"1499_CR6","unstructured":"Fu, C., Wu, X., Hu, Y., Huang, H., & He, R. (2019). Dual variational generation for low-shot heterogeneous face recognition. In NeurIPS."},{"key":"1499_CR7","doi-asserted-by":"crossref","unstructured":"Fu, J., Zheng, H., & Mei, T. (2017). Look closer to see better: Recurrent attention convolutional neural network for fine-grained image recognition. In CVPR.","DOI":"10.1109\/CVPR.2017.476"},{"key":"1499_CR8","doi-asserted-by":"crossref","unstructured":"Gao, L., Huang, D., Guo, Y., & Wang, Y. (2019). Pedestrian attribute recognition via hierarchical multi-task learning and relationship attention. In ACM MM.","DOI":"10.1145\/3343031.3351003"},{"key":"1499_CR9","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2014). Generative adversarial nets. In NeurIPS."},{"key":"1499_CR10","doi-asserted-by":"crossref","unstructured":"Guo, H., Zheng, K., Fan, X., Yu, H., & Wang, S. (2019). Visual attention consistency under image transforms for multi-label image classification. In CVPR.","DOI":"10.1109\/CVPR.2019.00082"},{"key":"1499_CR11","doi-asserted-by":"crossref","unstructured":"Han, K., Wang, Y., Shu, H., Liu, C., Xu, C., & Xu, C. (2019). Attribute aware pooling for pedestrian attribute recognition. In IJCAI.","DOI":"10.24963\/ijcai.2019\/341"},{"key":"1499_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask r-cnn. In ICCV.","DOI":"10.1109\/ICCV.2017.322"},{"key":"1499_CR13","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In CVPR.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1499_CR14","unstructured":"Hongyi, Z., Moustapha, C., Dauphin, Y. N., & Lopez-Paz, D. (2018). mixup: Beyond empirical risk minimization. In International conference on learning representations."},{"key":"1499_CR15","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., & Sun, G. (2018a). Squeeze-and-excitation networks. In CVPR (pp. 7132\u20137141).","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1499_CR16","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., & Sun, G. (2018b) . Squeeze-and-excitation networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1499_CR17","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der\u00a0Maaten, L., & Weinberger, K. Q. (2017). Densely connected convolutional networks. In CVPR (pp. 4700\u20134708).","DOI":"10.1109\/CVPR.2017.243"},{"key":"1499_CR18","unstructured":"Ioffe, S., & Szegedy, C. (2015) . Batch normalization: Accelerating deep network training by reducing internal covariate shift. In ICML."},{"key":"1499_CR19","unstructured":"Jia, J., Huang, H., Yang, W., Chen, X., & Huang, K. (2020). Rethinking of pedestrian attribute recognition: Realistic datasets with efficient method. arXiv preprint arXiv:2005.11909"},{"key":"1499_CR20","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., & Darrell, T. (2014). Caffe: Convolutional architecture for fast feature embedding. In ACM MM.","DOI":"10.1145\/2647868.2654889"},{"key":"1499_CR21","unstructured":"Kingma, D. P., & Ba, J. (2015). Adam: A method for stochastic optimization. In ICLR."},{"key":"1499_CR22","unstructured":"Kingma, D. P., & Welling, M. (2014). Auto-encoding variational bayes. In ICLR."},{"key":"1499_CR23","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In NeurIPS."},{"key":"1499_CR24","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P., et al. (1998). Gradient-based learning applied to document recognition. Proceedings of the IEEE, 86, 2278\u20132324.","journal-title":"Proceedings of the IEEE"},{"key":"1499_CR25","doi-asserted-by":"crossref","unstructured":"Li, D., Chen, X., & Huang, K. (2015). Multi-attribute learning for pedestrian attribute recognition in surveillance scenarios. In ACPR.","DOI":"10.1109\/ACPR.2015.7486476"},{"key":"1499_CR26","doi-asserted-by":"crossref","unstructured":"Li, D., Chen, X., Zhang, Z., & Huang, K. (2018a). Pose guided deep model for pedestrian attribute recognition in surveillance scenarios. In ICME.","DOI":"10.1109\/ICME.2018.8486604"},{"key":"1499_CR27","doi-asserted-by":"crossref","unstructured":"Li, D., Zhang, Z., Chen, X., & Huang, K. (2018b). A richly annotated pedestrian dataset for person retrieval in real surveillance scenarios. In IEEE TIP.","DOI":"10.1109\/TIP.2018.2878349"},{"key":"1499_CR28","doi-asserted-by":"crossref","unstructured":"Li, Q., Zhao, X., He, R., & Huang, K. (2019a). Pedestrian attribute recognition by joint visual-semantic reasoning and knowledge distillation. In IJCAI.","DOI":"10.24963\/ijcai.2019\/117"},{"key":"1499_CR29","doi-asserted-by":"crossref","unstructured":"Li, Q., Zhao, X., He, R., & Huang, K. (2019b). Visual-semantic graph reasoning for pedestrian attribute recognition. In AAAI.","DOI":"10.1609\/aaai.v33i01.33018634"},{"key":"1499_CR30","doi-asserted-by":"crossref","unstructured":"Li, W., Zhu, X., & Gong, S. (2018c). Harmonious attention network for person re-identification. In CVPR.","DOI":"10.1109\/CVPR.2018.00243"},{"key":"1499_CR31","doi-asserted-by":"crossref","unstructured":"Lim, J. J., Salakhutdinov, R. R., & Torralba, A. (2011). Transfer learning by borrowing examples for multiclass object detection. In NeurIPS.","DOI":"10.1109\/CVPR.2011.5995720"},{"key":"1499_CR32","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1016\/j.patcog.2019.06.006","volume":"95","author":"Y Lin","year":"2019","unstructured":"Lin, Y., Zheng, L., Zheng, Z., Wu, Y., Hu, Z., Yan, C., et al. (2019). Improving person re-identification by attribute and identity learning. Pattern Recognition, 95, 151\u2013161.","journal-title":"Pattern Recognition"},{"key":"1499_CR33","doi-asserted-by":"crossref","unstructured":"Liu, B., Wang, X., Dixit, M., Kwitt, R., & Vasconcelos, N. (2018a). Feature space transfer for data augmentation. In CVPR.","DOI":"10.1109\/CVPR.2018.00947"},{"key":"1499_CR34","unstructured":"Liu, P., Liu, X., Yan, J., & Shao, J.(2018b). Localization guided learning for pedestrian attribute recognition. In BMVC."},{"key":"1499_CR35","doi-asserted-by":"crossref","unstructured":"Liu, X., Zhao, H., Tian, M., Sheng, L., Shao, J., Yi, S., Yan, J., & Wang, X. (2017). Hydraplus-net: Attentive deep features for pedestrian analysis. In ICCV.","DOI":"10.1109\/ICCV.2017.46"},{"key":"1499_CR36","doi-asserted-by":"crossref","unstructured":"Liu, L., Ouyang, W., Wang, X., Fieguth, P., Chen, J., Liu, X., & Pietikainen, M. (2020). Deep learning for generic object detection: A survey. In IJCV.","DOI":"10.1007\/s11263-019-01247-4"},{"key":"1499_CR37","doi-asserted-by":"crossref","unstructured":"Sarafianos, N., Xu, X., & Kakadiaris, I. A. (2018). Deep imbalanced attribute classification using visual attention aggregation. In ECCV.","DOI":"10.1007\/978-3-030-01252-6_42"},{"key":"1499_CR38","unstructured":"Sarfraz, M. S., Schumann, A., Wang, Y., & Stiefelhagen, R. (2017). Deep view-sensitive pedestrian attribute inference in an end-to-end model. In BMVC."},{"key":"1499_CR39","unstructured":"Shifeng, Z., Longyin, W., Shi, H., Lei, Z., Lyu, S., & Li, S. Z. (2019). Single-shot scale-aware network for real-time face detection. In IJCV."},{"key":"1499_CR40","unstructured":"Shuzhe, W., Meina, K., Shan, S., & Chen, X. (2019). Hierarchical attention for part-aware face detection. In IJCV."},{"key":"1499_CR41","unstructured":"Simonyan, K., & Zisserman, A. (2015). Very deep convolutional networks for large-scale image recognition. In ICLR."},{"key":"1499_CR42","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Ioffe, S., Vanhoucke, V., & Alemi, A. (2017). Inception-v4, inception-resnet and the impact of residual connections on learning. In Proceedings of the AAAI conference on artificial intelligence (Vol.\u00a031).","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"1499_CR43","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., & Rabinovich, A. (2015). Going deeper with convolutions. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 1\u20139).","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1499_CR44","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., & Rabinovich, A. (2015). Going deeper with convolutions. In CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1499_CR45","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., & Wojna, Z. (2016). Rethinking the inception architecture for computer vision. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2818\u20132826).","DOI":"10.1109\/CVPR.2016.308"},{"key":"1499_CR46","doi-asserted-by":"crossref","unstructured":"Tan, Z., Wan, J., Lei, Z., Zhi, R., Guo, G., & Li, S. Z. (2018). Efficient group-n encoding and decoding for facial age estimation. In IEEE TPAMI.","DOI":"10.1109\/TPAMI.2017.2779808"},{"key":"1499_CR47","doi-asserted-by":"crossref","unstructured":"Tan, Z., Yang, Y., Wan, J., Guo, G., & Li, S. Z. (2019a). Deeply-learned hybrid representations for facial age estimation. In IJCAI (pp. 3548\u20133554).","DOI":"10.24963\/ijcai.2019\/492"},{"issue":"12","key":"1499_CR48","first-page":"6126","volume":"28","author":"Z Tan","year":"2019","unstructured":"Tan, Z., Yang, Y., Wan, J., Hang, H., Guo, G., & Li, S. Z. (2019b). Attention-based pedestrian attribute analysis. IEEE TIP, 28(12), 6126\u20136140.","journal-title":"IEEE TIP"},{"key":"1499_CR49","doi-asserted-by":"crossref","unstructured":"Tang, C., Sheng, L., Zhang, Z., & Hu, X. (2019c). Improving pedestrian attribute recognition with weakly-supervised multi-scale attribute-specific localization. In ICCV.","DOI":"10.1109\/ICCV.2019.00510"},{"key":"1499_CR50","doi-asserted-by":"crossref","unstructured":"Wang, J., Zhu, X., Gong, S., & Li, W. (2017). Attribute recognition by joint recurrent learning of context and correlation. In ICCV.","DOI":"10.1109\/ICCV.2017.65"},{"key":"1499_CR51","doi-asserted-by":"crossref","unstructured":"Wang, Y., Gan, W., Wu, W., & Yan, J. (2019). Dynamic curriculum learning for imbalanced data classification. In ICCV.","DOI":"10.1109\/ICCV.2019.00512"},{"key":"1499_CR52","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J. Y., & So\u00a0Kweon, I. (2018). Cbam: Convolutional block attention module. In ECCV.","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"1499_CR53","doi-asserted-by":"crossref","unstructured":"Wu, M., Huang, D., Guo, Y., & Wang, Y. (2020). Distraction-aware feature learning for human attribute recognition via coarse-to-fine attention mechanism. In AAAI.","DOI":"10.1609\/aaai.v34i07.6925"},{"key":"1499_CR54","doi-asserted-by":"crossref","unstructured":"Xiang, L., Jin, X., Ding, G., Han, J., & Li, L. (2019). Incremental few-shot learning for pedestrian attribute recognition. In IJCAI.","DOI":"10.24963\/ijcai.2019\/543"},{"key":"1499_CR55","unstructured":"Xiangyu, Z., Hao, L., & Gong, S. (2020). Scalable person re-identification by harmonious attention. In IJCV."},{"key":"1499_CR56","unstructured":"Yu, F., & Koltun, V. (2016). Multi-scale context aggregation by dilated convolutions. In ICLR."},{"key":"1499_CR57","doi-asserted-by":"crossref","unstructured":"Zeng, H., Ai, H., Zhuang, Z., & Chen, L. (2020). Multi-task learning via co-attentive sharing for pedestrian attribute recognition. In ICME.","DOI":"10.1109\/ICME46284.2020.9102757"},{"key":"1499_CR58","unstructured":"Zhang, H., Wu, C., Zhang, Z., Zhu, Y., Lin, H., Zhang, Z., Sun, Y., He, T., Mueller, J., Manmatha, R., & Li, M. (2020). Resnest: Split-attention networks. arXiv preprint arXiv:2004.08955"},{"key":"1499_CR59","unstructured":"Zhang, J., Ren, P., & Li, J. (2020) . Deep template matching for pedestrian attribute recognition with the auxiliary supervision of attribute-wise keypoints. arXiv preprint arXiv:2011.06798"},{"key":"1499_CR60","doi-asserted-by":"crossref","unstructured":"Zhao, X., Sang, L., Ding, G., Guo, Y., & Jin, X. (2018). Grouping attribute recognition for pedestrian with joint recurrent learning. In IJCAI.","DOI":"10.24963\/ijcai.2018\/441"},{"key":"1499_CR61","doi-asserted-by":"crossref","unstructured":"Zhao, X., Sang, L., Ding, G., Han, J., Di, N., & Yan, C. (2019). Recurrent attention model for pedestrian attribute recognition. In: AAAI.","DOI":"10.1609\/aaai.v33i01.33019275"},{"key":"1499_CR62","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Zheng, L., & Yang, Y. (2017). Unlabeled samples generated by gan improve the person re-identification baseline in vitro. In ICCV.","DOI":"10.1109\/ICCV.2017.405"},{"key":"1499_CR63","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Zheng, L., Kang, G., Li, S., & Yang, Y. (2020). Random erasing data augmentation. In Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v34i07.7000"},{"key":"1499_CR64","doi-asserted-by":"crossref","unstructured":"Zhu, J., Liao, S., Lei, Z., Yi, D., & Li, S. Z. (2013). Pedestrian attribute classification in surveillance: Database and evaluation. In ICCVW.","DOI":"10.1109\/ICCVW.2013.51"},{"key":"1499_CR65","doi-asserted-by":"crossref","unstructured":"Zhu, X., Liu, H., Lei, Z., Shi, H., Yang, F., Yi, D., Qi, G., & Li, S. Z. (2019). Large-scale bisample learning on id versus spot face recognition. In IJCV.","DOI":"10.1007\/s11263-019-01162-8"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01499-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-021-01499-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01499-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,4]],"date-time":"2023-01-04T20:13:46Z","timestamp":1672863226000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-021-01499-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,18]]},"references-count":65,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2021,10]]}},"alternative-id":["1499"],"URL":"https:\/\/doi.org\/10.1007\/s11263-021-01499-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,7,18]]},"assertion":[{"value":"27 June 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 June 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 July 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}