{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T02:31:35Z","timestamp":1771468295043,"version":"3.50.1"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030586096","type":"print"},{"value":"9783030586102","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-58610-2_11","type":"book-chapter","created":{"date-parts":[[2020,10,6]],"date-time":"2020-10-06T13:02:49Z","timestamp":1601989369000},"page":"174-190","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":23,"title":["Online Multi-modal Person Search in Videos"],"prefix":"10.1007","author":[{"given":"Jiangyue","family":"Xia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anyi","family":"Rao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingqiu","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linning","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiangtao","family":"Wen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dahua","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,10,7]]},"reference":[{"key":"11_CR1","doi-asserted-by":"crossref","unstructured":"Arandjelovic, O., Zisserman, A.: Automatic face recognition for film character retrieval in feature-length films. In: 2005 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 860\u2013867 (2005)","DOI":"10.1109\/CVPR.2005.81"},{"key":"11_CR2","unstructured":"Chung, J.S.: Naver at ActivityNet challenge 2019-task B active speaker detection (AVA). arXiv preprint arXiv:1906.10555 (2019)"},{"key":"11_CR3","doi-asserted-by":"crossref","unstructured":"Cour, T., Sapp, B., Nagle, A., Taskar, B.: Talking pictures: temporal grouping and dialog-supervised person recognition. In: 2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1014\u20131021 (2010)","DOI":"10.1109\/CVPR.2010.5540106"},{"issue":"5","key":"11_CR4","doi-asserted-by":"publisher","first-page":"840","DOI":"10.1109\/TMM.2005.854464","volume":"7","author":"E Erzin","year":"2005","unstructured":"Erzin, E., Yemez, Y., Tekalp, A.M.: Multimodal speaker identification using an adaptive classifier cascade based on modality reliability. IEEE Trans. Multimedia 7(5), 840\u2013852 (2005)","journal-title":"IEEE Trans. Multimedia"},{"key":"11_CR5","doi-asserted-by":"crossref","unstructured":"Everingham, M., Sivic, J., Zisserman, A.: \u201cHello! my name is... Buffy\u201d-automatic naming of characters in TV video. In: 2006 British Machine Vision Conference (BMVC), pp. 899\u2013908 (2006)","DOI":"10.5244\/C.20.92"},{"key":"11_CR6","doi-asserted-by":"crossref","unstructured":"Farenzena, M., Bazzani, L., Perina, A., Murino, V., Cristani, M.: Person re-identification by symmetry-driven accumulation of local features. In: 2010 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2360\u20132367 (2010)","DOI":"10.1109\/CVPR.2010.5539926"},{"key":"11_CR7","doi-asserted-by":"crossref","unstructured":"Feng, L., Li, Z., Kuang, Z., Zhang, W.: Extractive video summarizer with memory augmented neural networks. In: 2018 ACM International Conference on Multimedia (MM), pp. 976\u2013983 (2018)","DOI":"10.1145\/3240508.3240651"},{"key":"11_CR8","doi-asserted-by":"crossref","unstructured":"Gheissari, N., Sebastian, T.B., Hartley, R.: Person reidentification using spatiotemporal appearance. In: 2006 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1528\u20131535 (2006)","DOI":"10.1109\/CVPR.2006.223"},{"key":"11_CR9","unstructured":"Graves, A., Wayne, G., Danihelka, I.: Neural turing machines. arXiv preprint arXiv:1410.5401 (2014)"},{"key":"11_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1007\/978-3-319-46487-9_6","volume-title":"Computer Vision \u2013 ECCV 2016","author":"Y Guo","year":"2016","unstructured":"Guo, Y., Zhang, L., Hu, Y., He, X., Gao, J.: MS-Celeb-1M: a dataset and benchmark for large-scale face recognition. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9907, pp. 87\u2013102. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46487-9_6"},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Haurilet, M., Tapaswi, M., Al-Halah, Z., Stiefelhagen, R.: Naming TV characters by watching and analyzing dialogs. In: 2016 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1\u20139 (2016)","DOI":"10.1109\/WACV.2016.7477560"},{"key":"11_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"11_CR13","doi-asserted-by":"crossref","unstructured":"Hu, D., Li, X., Lu, X.: Temporal multimodal learning in audiovisual speech recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3574\u20133582 (2016)","DOI":"10.1109\/CVPR.2016.389"},{"key":"11_CR14","doi-asserted-by":"crossref","unstructured":"Hu, Y., Ren, J.S., Dai, J., Yuan, C., Xu, L., Wang, W.: Deep multimodal speaker naming. In: 2015 ACM International Conference on Multimedia (MM), pp. 1107\u20131110 (2015)","DOI":"10.1145\/2733373.2806293"},{"key":"11_CR15","doi-asserted-by":"crossref","unstructured":"Huang, Q., Xiong, Y., Lin, D.: Unifying identification and context learning for person recognition. In: 2018 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2217\u20132225 (2018)","DOI":"10.1109\/CVPR.2018.00236"},{"key":"11_CR16","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1007\/978-3-030-01261-8_26","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Q Huang","year":"2018","unstructured":"Huang, Q., Liu, W., Lin, D.: Person search in videos with one portrait through visual and temporal links. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11217, pp. 437\u2013454. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01261-8_26"},{"key":"11_CR17","doi-asserted-by":"crossref","unstructured":"Huang, Q., Xiong, Y., Rao, A., Wang, J., Lin, D.: MovieNet: a holistic dataset for movie understanding. In: 2020 European Conference on Computer Vision (ECCV) (2020)","DOI":"10.1007\/978-3-030-58548-8_41"},{"key":"11_CR18","unstructured":"Li, D., Kadav, A.: Adaptive memory networks. In: 2018 International Conference on Learning Representations Workshop (ICLRW) (2018)"},{"key":"11_CR19","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1007\/978-3-642-15549-9_18","volume-title":"Computer Vision \u2013 ECCV 2010","author":"D Lin","year":"2010","unstructured":"Lin, D., Kapoor, A., Hua, G., Baker, S.: Joint people, event, and location recognition in personal photo collections using cross-domain context. In: Daniilidis, K., Maragos, P., Paragios, N. (eds.) ECCV 2010. LNCS, vol. 6311, pp. 243\u2013256. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15549-9_18"},{"key":"11_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, J., Gong, S., Lu, H., Tao, D.: Deep reinforcement active learning for human-in-the-loop person re-identification. In: 2019 IEEE International Conference on Computer Vision (ICCV), pp. 6121\u20136130 (2019)","DOI":"10.1109\/ICCV.2019.00622"},{"key":"11_CR21","unstructured":"Logan, B.: Mel frequency cepstral coefficients for music modeling. In: 2000 International Symposium on Music Information Retrieval (ISMIR) (2000)"},{"key":"11_CR22","unstructured":"Loy, C.C., et al.: Wider face and pedestrian challenge 2018: methods and results. arXiv preprint arXiv:1902.06854 (2019)"},{"issue":"Nov","key":"11_CR23","first-page":"2579","volume":"9","author":"LVD Maaten","year":"2008","unstructured":"Maaten, L.V.D., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(Nov), 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"11_CR24","doi-asserted-by":"crossref","unstructured":"Na, S., Lee, S., Kim, J., Kim, G.: A read-write memory network for movie story understanding. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 677\u2013685 (2017)","DOI":"10.1109\/ICCV.2017.80"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Zisserman, A.: From Benedict Cumberbatch to Sherlock Holmes: character identification in TV series without a script. In: 2017 British Machine Vision Conference (BMVC), pp. 107.1\u2013107.13 (2017)","DOI":"10.5244\/C.31.107"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"Ouyang, D., Shao, J., Zhang, Y., Yang, Y., Shen, H.T.: Video-based person re-identification via self-paced learning and deep reinforcement learning framework. In: 2018 ACM International Conference on Multimedia (MM), pp. 1562\u20131570 (2018)","DOI":"10.1145\/3240508.3240622"},{"key":"11_CR27","doi-asserted-by":"crossref","unstructured":"Rao, A., et al.: A unified framework for shot type classification based on subject centric lens. In: 2020 European Conference on Computer Vision (ECCV) (2020)","DOI":"10.1007\/978-3-030-58621-8_2"},{"key":"11_CR28","doi-asserted-by":"crossref","unstructured":"Rao, A., et al.: A local-to-global approach to multi-modal movie scene segmentation. In: 2020 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10146\u201310155 (2020)","DOI":"10.1109\/CVPR42600.2020.01016"},{"key":"11_CR29","doi-asserted-by":"crossref","unstructured":"Rao, Y., Lu, J., Zhou, J.: Attention-aware deep reinforcement learning for video face recognition. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 3951\u20133960 (2017)","DOI":"10.1109\/ICCV.2017.424"},{"key":"11_CR30","doi-asserted-by":"crossref","unstructured":"Ren, J.S.J., et al.: Look, listen and learn - a multimodal LSTM for speaker identification. In: 2016 AAAI Conference on Artificial Intelligence (AAAI), pp. 3581\u20133587 (2016)","DOI":"10.1609\/aaai.v30i1.10471"},{"key":"11_CR31","doi-asserted-by":"crossref","unstructured":"Roth, J., et al.: AVA-active speaker: an audio-visual dataset for active speaker detection. arXiv preprint arXiv:1901.01342 (2019)","DOI":"10.1109\/ICASSP40776.2020.9053900"},{"issue":"3","key":"11_CR32","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge. Int. J. Comput. Vis. 115(3), 211\u2013252 (2015)","journal-title":"Int. J. Comput. Vis."},{"key":"11_CR33","unstructured":"Shen, Y., Tan, S., Hosseini, A., Lin, Z., Sordoni, A., Courville, A.C.: Ordered memory. In: Advances in Neural Information Processing Systems, pp. 5037\u20135048 (2019)"},{"key":"11_CR34","doi-asserted-by":"crossref","unstructured":"Sivic, J., Everingham, M., Zisserman, A.: \u201cWho are you?\u201d - learning person specific classifiers from video. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1145\u20131152 (2009)","DOI":"10.1109\/CVPR.2009.5206513"},{"key":"11_CR35","unstructured":"Sukhbaatar, S., Szlam, A., Weston, J., Fergus, R.: End-to-end memory networks. In: Advances in Neural Information Processing Systems, pp. 2440\u20132448 (2015)"},{"issue":"5","key":"11_CR36","doi-asserted-by":"publisher","first-page":"1054","DOI":"10.1109\/TNN.1998.712192","volume":"9","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement learning: an introduction. IEEE Trans. Neural Netw. 9(5), 1054\u20131054 (1998)","journal-title":"IEEE Trans. Neural Netw."},{"key":"11_CR37","doi-asserted-by":"crossref","unstructured":"Wang, J., Wang, W., Huang, Y., Wang, L., Tan, T.: Hierarchical memory modelling for video captioning. In: 2018 ACM International Conference on Multimedia (MM), pp. 63\u201371 (2018)","DOI":"10.1145\/3240508.3240538"},{"key":"11_CR38","doi-asserted-by":"crossref","unstructured":"Wang, J., Wang, W., Wang, Z., Wang, L., Feng, D., Tan, T.: Stacked memory network for video summarization. In: 2019 ACM International Conference on Multimedia (MM), pp. 836\u2013844 (2019)","DOI":"10.1145\/3343031.3350992"},{"key":"11_CR39","unstructured":"Weston, J., Chopra, S., Bordes, A.: Memory networks. In: 2015 International Conference on Learning Representations (ICLR) (2015)"},{"key":"11_CR40","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1007\/978-3-030-01240-3_10","volume-title":"Computer Vision \u2013 ECCV 2018","author":"T Yang","year":"2018","unstructured":"Yang, T., Chan, A.B.: Learning dynamic memory networks for object tracking. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11213, pp. 153\u2013169. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01240-3_10"},{"key":"11_CR41","unstructured":"Zajdel, W., Zivkovic, Z., Krose, B.J.A.: Keeping track of humans: have I seen this person before? In: 2005 IEEE International Conference on Robotics and Automation (ICRA), pp. 2081\u20132086 (2005)"},{"issue":"10","key":"11_CR42","doi-asserted-by":"publisher","first-page":"1499","DOI":"10.1109\/LSP.2016.2603342","volume":"23","author":"K Zhang","year":"2016","unstructured":"Zhang, K., Zhang, Z., Li, Z., Qiao, Y.: Joint face detection and alignment using multitask cascaded convolutional networks. IEEE Signal Process. Lett. 23(10), 1499\u20131503 (2016)","journal-title":"IEEE Signal Process. Lett."},{"key":"11_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, N., Paluri, M., Taigman, Y., Fergus, R., Bourdev, L.: Beyond frontal faces: improving person recognition using multiple cues. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4804\u20134813 (2015)","DOI":"10.1109\/CVPR.2015.7299113"},{"issue":"12","key":"11_CR44","doi-asserted-by":"publisher","first-page":"3847","DOI":"10.1109\/TNNLS.2019.2899588","volume":"30","author":"W Zhang","year":"2019","unstructured":"Zhang, W., He, X., Lu, W., Qiao, H., Li, Y.: Feature aggregation with reinforcement learning for video-based person re-identification. IEEE Trans. Neural Netw. Learn. Syst. 30(12), 3847\u20133852 (2019)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"11_CR45","doi-asserted-by":"crossref","unstructured":"Zheng, L., et al.: MARS: a video benchmark for large-scale person re-identification. In: 2016 European Conference on Computer Vision (ECCV), pp. 868\u2013884 (2016)","DOI":"10.1007\/978-3-319-46466-4_52"},{"key":"11_CR46","unstructured":"Zhou, D., Bousquet, O., Lal, T.N., Weston, J., Sch\u00f6lkopf, B.: Learning with local and global consistency. In: Advances in Neural Information Processing Systems, pp. 321\u2013328 (2003)"},{"key":"11_CR47","doi-asserted-by":"crossref","unstructured":"Zhou, H., Liu, Z., Xu, X., Luo, P., Wang, X.: Vision-infused deep audio inpainting. In: 2019 IEEE International Conference on Computer Vision (ICCV), pp. 283\u2013292 (2019)","DOI":"10.1109\/ICCV.2019.00037"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2020"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-58610-2_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,6]],"date-time":"2024-10-06T00:28:34Z","timestamp":1728174514000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-58610-2_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030586096","9783030586102"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-58610-2_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"7 October 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Glasgow","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 August 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2020.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OpenReview","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5025","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1360","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"27% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually due to the COVID-19 pandemic. From the ECCV Workshops 249 full papers, 18 short papers, and 21 further contributions were published out of a total of 467 submissions.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}