{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T15:54:05Z","timestamp":1782402845784,"version":"3.54.5"},"publisher-location":"Cham","reference-count":55,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031197802","type":"print"},{"value":"9783031197819","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-19781-9_32","type":"book-chapter","created":{"date-parts":[[2022,10,22]],"date-time":"2022-10-22T12:12:59Z","timestamp":1666440779000},"page":"549-566","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["CAViT: Contextual Alignment Vision Transformer for\u00a0Video Object Re-identification"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7877-5728","authenticated-orcid":false,"given":"Jinlin","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lingxiao","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1633-7575","authenticated-orcid":false,"given":"Wu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0559-5464","authenticated-orcid":false,"given":"Yang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0791-189X","authenticated-orcid":false,"given":"Zhen","family":"Lei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5990-7307","authenticated-orcid":false,"given":"Tao","family":"Mei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2961-8096","authenticated-orcid":false,"given":"Stan Z.","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,10,23]]},"reference":[{"key":"32_CR1","doi-asserted-by":"crossref","unstructured":"Aich, A., Zheng, M., Karanam, S., Chen, T., Roy-Chowdhury, A.K., Wu, Z.: Spatio-temporal representation factorization for video-based person re-identification, In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00022"},{"key":"32_CR2","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? arXiv preprint arXiv:2102.05095 (2021)"},{"key":"32_CR3","unstructured":"Bochkovskiy, A., Wang, C.Y., Liao, H.Y.M.: Yolov4: Optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934 (2020)"},{"key":"32_CR4","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"32_CR5","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"32_CR6","doi-asserted-by":"crossref","unstructured":"Chen, C.F., Fan, Q., Panda, R.: Crossvit: Cross-attention multi-scale vision transformer for image classification. arXiv preprint arXiv:2103.14899 (2021)","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"32_CR7","doi-asserted-by":"crossref","unstructured":"Chen, G., Rao, Y., Lu, J., Zhou, J.: Temporal coherence or temporal motion: Which is more critical for video-based person re-identification? In: ECCV (2020)","DOI":"10.1007\/978-3-030-58598-3_39"},{"key":"32_CR8","doi-asserted-by":"crossref","unstructured":"Dehghan, A., Modiri Assari, S., Shah, M.: Gmmcp tracker: Globally optimal generalized maximum multi clique problem for multiple object tracking. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7299036"},{"key":"32_CR9","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"32_CR10","doi-asserted-by":"crossref","unstructured":"Eom, C., Lee, G., Lee, J., Ham, B.: Video-based person re-identification with spatial and temporal memory networks. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01182"},{"key":"32_CR11","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"32_CR12","doi-asserted-by":"crossref","unstructured":"Felzenszwalb, P.F., Girshick, R.B., McAllester, D., Ramanan, D.: Object detection with discriminatively trained part-based models. IEEE TPAMI (2009)","DOI":"10.1109\/TPAMI.2009.167"},{"key":"32_CR13","doi-asserted-by":"crossref","unstructured":"Gu, X., Chang, H., Ma, B., Zhang, H., Chen, X.: Appearance-preserving 3d convolution for video-based person re-identification. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58536-5_14"},{"key":"32_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"32_CR15","unstructured":"He, L., Liao, X., Liu, W., Liu, X., Cheng, P., Mei, T.: Fastreid: a pytorch toolbox for real-world person re-identification. arXiv preprint arXiv:2006.02631 (2020)"},{"key":"32_CR16","doi-asserted-by":"crossref","unstructured":"He, S., Luo, H., Wang, P., Wang, F., Li, H., Jiang, W.: Transreid: Transformer-based object re-identification. arXiv preprint arXiv:2102.04378 (2021)","DOI":"10.1109\/ICCV48922.2021.01474"},{"key":"32_CR17","doi-asserted-by":"crossref","unstructured":"He, T., Jin, X., Shen, X., Huang, J., Chen, Z., Hua, X.S.: Dense interaction learning for video-based person re-identification supplementary materials. Identities (2021)","DOI":"10.1109\/ICCV48922.2021.00152"},{"key":"32_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1007\/978-3-642-21227-7_9","volume-title":"Image Analysis","author":"M Hirzer","year":"2011","unstructured":"Hirzer, M., Beleznai, C., Roth, P.M., Bischof, H.: Person re-identification by descriptive and discriminative classification. In: Heyden, A., Kahl, F. (eds.) SCIA 2011. LNCS, vol. 6688, pp. 91\u2013102. Springer, Heidelberg (2011). https:\/\/doi.org\/10.1007\/978-3-642-21227-7_9"},{"key":"32_CR19","doi-asserted-by":"crossref","unstructured":"Hou, R., Chang, H., Ma, B., Huang, R., Shan, S.: Bicnet-tks: Learning efficient spatial-temporal representation for video person re-identification. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00205"},{"key":"32_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"388","DOI":"10.1007\/978-3-030-58595-2_24","volume-title":"Computer Vision \u2013 ECCV 2020","author":"R Hou","year":"2020","unstructured":"Hou, R., Chang, H., Ma, B., Shan, S., Chen, X.: Temporal complementary learning for video person re-identification. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12370, pp. 388\u2013405. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58595-2_24"},{"key":"32_CR21","doi-asserted-by":"crossref","unstructured":"Hou, R., Ma, B., Chang, H., Gu, X., Shan, S., Chen, X.: Iaunet: Global context-aware feature learning for person reidentification. IEEE TNNLS (2020)","DOI":"10.1109\/TNNLS.2020.3017939"},{"key":"32_CR22","doi-asserted-by":"crossref","unstructured":"Hou, R., Ma, B., Chang, H., Gu, X., Shan, S., Chen, X.: Feature completion for occluded person re-identification. IEEE TPAMI (2021)","DOI":"10.1109\/TPAMI.2021.3079910"},{"key":"32_CR23","unstructured":"Zhao, J., Qi, F., G.R., Xu, L.: Vveri-901: Video vehicle re-identification dataset (2020). https:\/\/www.graviti.cn\/open-datasets\/VVeRI901\u2019"},{"key":"32_CR24","doi-asserted-by":"crossref","unstructured":"Li, C., Zhong, Q., Xie, D., Pu, S.: Collaborative spatiotemporal feature learning for video action recognition. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00806"},{"key":"32_CR25","doi-asserted-by":"crossref","unstructured":"Li, J., Wang, J., Tian, Q., Gao, W., Zhang, S.: Global-local temporal representations for video person re-identification. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00406"},{"key":"32_CR26","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, S., Huang, T.: Multi-scale 3D convolution network for video based person re-identification. In: AAAI (2019)","DOI":"10.1609\/aaai.v33i01.33018618"},{"key":"32_CR27","doi-asserted-by":"crossref","unstructured":"Li, S., Bak, S., Carr, P., Wang, X.: Diversity regularized spatiotemporal attention for video-based person re-identification. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00046"},{"key":"32_CR28","unstructured":"Li, S.Z.: Markov random field modeling in image analysis. Springer Science & Business Media (2009)"},{"key":"32_CR29","doi-asserted-by":"crossref","unstructured":"Li, X., Zhou, W., Zhou, Y., Li, H.: Relation-guided spatial attention and temporal refinement for video-based person re-identification. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i07.6807"},{"key":"32_CR30","doi-asserted-by":"crossref","unstructured":"Li, Y., He, J., Zhang, T., Liu, X., Zhang, Y., Wu, F.: Diverse part discovery: Occluded person re-identification with part-aware transformer. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00292"},{"key":"32_CR31","doi-asserted-by":"crossref","unstructured":"Liao, S., Shao, L.: Transformer-based deep image matching for generalizable person re-identification. NeurIPS Workshops (2021)","DOI":"10.1109\/CVPR52688.2022.00721"},{"key":"32_CR32","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S.: Tsm: Temporal shift module for efficient video understanding. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"32_CR33","unstructured":"Liu, C.T., Wu, C.W., Wang, Y.C.F., Chien, S.Y.: Spatially and temporally efficient non-local attention network for video-based person re-identification. arXiv preprint arXiv:1908.01683 (2019)"},{"key":"32_CR34","doi-asserted-by":"crossref","unstructured":"Liu, J., Zha, Z.J., Wu, W., Zheng, K., Sun, Q.: Spatial-temporal correlation and topology learning for person re-identification in videos. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00435"},{"key":"32_CR35","doi-asserted-by":"crossref","unstructured":"Liu, X., Zhang, P., Yu, C., Lu, H., Yang, X.: Watching you: Global-guided reciprocal learning for video-based person re-identification. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01313"},{"key":"32_CR36","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: Hierarchical vision transformer using shifted windows. ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"32_CR37","unstructured":"Liu, Z., et al.: Video swin transformer. arXiv preprint arXiv:2106.13230 (2021)"},{"key":"32_CR38","doi-asserted-by":"crossref","unstructured":"Luo, H., Gu, Y., Liao, X., Lai, S., Jiang, W.: Bag of tricks and a strong baseline for deep person re-identification. In: CVPR Workshops (2019)","DOI":"10.1109\/CVPRW.2019.00190"},{"key":"32_CR39","doi-asserted-by":"crossref","unstructured":"Pathak, P., Eshratifar, A.E., Gormish, M.: Video person re-id: Fantastic techniques and where to find them. arXiv preprint arXiv:1912.05295 (2019)","DOI":"10.1609\/aaai.v34i10.7219"},{"key":"32_CR40","doi-asserted-by":"crossref","unstructured":"Qiu, Z., Yao, T., Mei, T.: Learning spatio-temporal representation with pseudo-3d residual networks. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.590"},{"key":"32_CR41","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"32_CR42","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"688","DOI":"10.1007\/978-3-319-10593-2_45","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T Wang","year":"2014","unstructured":"Wang, T., Gong, S., Zhu, X., Wang, S.: Person re-identification by video ranking. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8692, pp. 688\u2013703. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10593-2_45"},{"key":"32_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Y., Zhang, P., Gao, S., Geng, X., Lu, H., Wang, D.: Pyramid spatial-temporal aggregation for video-based person re-identification. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01181"},{"key":"32_CR44","unstructured":"Weng, X., Kitani, K.: Learning spatio-temporal features with two-stream deep 3d cnns for lipreading. arXiv preprint arXiv:1905.02540 (2019)"},{"key":"32_CR45","doi-asserted-by":"crossref","unstructured":"Wu, Y., et al.: Adaptive graph representation learning for video person re-identification. IEEE TIP (2020)","DOI":"10.1109\/TIP.2020.3001693"},{"key":"32_CR46","doi-asserted-by":"crossref","unstructured":"Yan, Y., et al.: Learning multi-granular hypergraphs for video-based person re-identification. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00297"},{"key":"32_CR47","doi-asserted-by":"crossref","unstructured":"Yang, J., Zheng, W.S., Yang, Q., Chen, Y.C., Tian, Q.: Spatial-temporal graph convolutional network for video-based person re-identification. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00335"},{"key":"32_CR48","unstructured":"Zhang, H., et al.: Resnest: Split-attention networks. arXiv preprint arXiv:2004.08955 (2020)"},{"key":"32_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, H., Hao, Y., Ngo, C.W.: Token shift transformer for video classification. In: ACM MM (2021)","DOI":"10.1145\/3474085.3475272"},{"key":"32_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Lan, C., Zeng, W., Chen, Z.: Multi-granularity reference-aided attentive feature aggregation for video-based person re-identification. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01042"},{"key":"32_CR51","doi-asserted-by":"crossref","unstructured":"Zhao, J., Qi, F., Ren, G., Xu, L.: Phd learning: Learning with pompeiu-hausdorff distances for video-based vehicle re-identification. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00226"},{"key":"32_CR52","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Shen, X., Jin, Z., Lu, H., Hua, X.s.: Attribute-driven feature disentangling and temporal aggregation for video person re-identification. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00505"},{"key":"32_CR53","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"868","DOI":"10.1007\/978-3-319-46466-4_52","volume-title":"Computer Vision \u2013 ECCV 2016","author":"L Zheng","year":"2016","unstructured":"Zheng, L., et al.: MARS: a video benchmark for large-scale Person re-identification. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9910, pp. 868\u2013884. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46466-4_52"},{"key":"32_CR54","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Huang, Y., Wang, W., Wang, L., Tan, T.: See the forest for the trees: Joint spatial and temporal recurrent neural networks for video-based person re-identification. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.717"},{"key":"32_CR55","unstructured":"Zhu, K., et al.: Aaformer: Auto-aligned transformer for person re-identification. arXiv preprint arXiv:2104.00921 (2021)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-19781-9_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,12]],"date-time":"2024-03-12T16:42:30Z","timestamp":1710261750000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-19781-9_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031197802","9783031197819"],"references-count":55,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-19781-9_32","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"23 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}