{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T17:27:14Z","timestamp":1783099634529,"version":"3.54.6"},"publisher-location":"Cham","reference-count":94,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031250842","type":"print"},{"value":"9783031250859","type":"electronic"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-25085-9_19","type":"book-chapter","created":{"date-parts":[[2023,2,11]],"date-time":"2023-02-11T09:12:42Z","timestamp":1676106762000},"page":"326-346","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["ModSelect: Automatic Modality Selection for\u00a0Synthetic-to-Real Domain Generalization"],"prefix":"10.1007","author":[{"given":"Zdravko","family":"Marinov","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alina","family":"Roitberg","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Schneider","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rainer","family":"Stiefelhagen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,2,12]]},"reference":[{"issue":"3","key":"19_CR1","doi-asserted-by":"publisher","first-page":"1445","DOI":"10.1109\/JSEN.2019.2947446","volume":"20","author":"Z Ahmad","year":"2019","unstructured":"Ahmad, Z., Khan, N.: Human action recognition using deep multilevel multimodal ($${M}^{2}$$) fusion of depth and inertial sensors. IEEE Sens. J. 20(3), 1445\u20131455 (2019)","journal-title":"IEEE Sens. J."},{"issue":"3","key":"19_CR2","doi-asserted-by":"publisher","first-page":"3623","DOI":"10.1109\/JSEN.2020.3028561","volume":"21","author":"Z Ahmad","year":"2020","unstructured":"Ahmad, Z., Khan, N.: CNN-based multistage gated average fusion (MGAF) for human action recognition using depth and inertial sensors. IEEE Sens. J. 21(3), 3623\u20133634 (2020)","journal-title":"IEEE Sens. J."},{"key":"19_CR3","first-page":"25","volume":"33","author":"JB Alayrac","year":"2020","unstructured":"Alayrac, J.B., et al.: Self-supervised multimodal versatile networks. Adv. Neural Inf. Process. Syst. 33, 25\u201337 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"19_CR4","first-page":"1","volume":"33","author":"H Alwassel","year":"2020","unstructured":"Alwassel, H., Mahajan, D., Korbar, B., Torresani, L., Ghanem, B., Tran, D.: Self-supervised learning by cross-modal audio-video clustering. Adv. Neural Inf. Process. Syst. 33, 1\u201313 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"19_CR5","doi-asserted-by":"crossref","unstructured":"Ardianto, S., Hang, H.M.: Multi-view and multi-modal action recognition with learned fusion. In: 2018 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), pp. 1601\u20131604. IEEE (2018)","DOI":"10.23919\/APSIPA.2018.8659539"},{"issue":"6","key":"19_CR6","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/s00530-010-0182-0","volume":"16","author":"PK Atrey","year":"2010","unstructured":"Atrey, P.K., Hossain, M.A., El Saddik, A., Kankanhalli, M.S.: Multimodal fusion for multimedia analysis: a survey. Multimedia Syst. 16(6), 345\u2013379 (2010)","journal-title":"Multimedia Syst."},{"key":"19_CR7","doi-asserted-by":"crossref","unstructured":"Baradel, F., Wolf, C., Mille, J.: Human action recognition: pose-based attention draws focus to hands. In: IEEE International Conference on Computer Vision Workshops, pp. 604\u2013613 (2017)","DOI":"10.1109\/ICCVW.2017.77"},{"key":"19_CR8","unstructured":"Black, D., et al.: The theory of committees and elections (1958)"},{"issue":"14","key":"19_CR9","doi-asserted-by":"publisher","first-page":"e49","DOI":"10.1093\/bioinformatics\/btl242","volume":"22","author":"KM Borgwardt","year":"2006","unstructured":"Borgwardt, K.M., Gretton, A., Rasch, M.J., Kriegel, H.P., Sch\u00f6lkopf, B., Smola, A.J.: Integrating structured biological data by kernel maximum mean discrepancy. Bioinformatics 22(14), e49\u2013e57 (2006)","journal-title":"Bioinformatics"},{"issue":"2","key":"19_CR10","doi-asserted-by":"publisher","first-page":"413","DOI":"10.1109\/TPAMI.2018.2880750","volume":"42","author":"PP Busto","year":"2018","unstructured":"Busto, P.P., Iqbal, A., Gall, J.: Open set domain adaptation for image and action recognition. IEEE Trans. Pattern Anal. Mach. Intell. 42(2), 413\u2013429 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"19_CR11","doi-asserted-by":"crossref","unstructured":"Caba Heilbron, F., Escorcia, V., Ghanem, B., Carlos Niebles, J.: Activitynet: a large-scale video benchmark for human activity understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 961\u2013970 (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"19_CR12","doi-asserted-by":"crossref","unstructured":"Cai, J., Jiang, N., Han, X., Jia, K., Lu, J.: Jolo-gcn: mining joint-centered light-weight information for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2735\u20132744 (2021)","DOI":"10.1109\/WACV48630.2021.00278"},{"key":"19_CR13","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"issue":"6","key":"19_CR14","doi-asserted-by":"publisher","first-page":"633","DOI":"10.1016\/j.cviu.2013.01.013","volume":"117","author":"JM Chaquet","year":"2013","unstructured":"Chaquet, J.M., Carmona, E.J., Fern\u00e1ndez-Caballero, A.: A survey of video datasets for human action and activity recognition. Comput. Vision Image Underst. 117(6), 633\u2013659 (2013)","journal-title":"Comput. Vision Image Underst."},{"key":"19_CR15","doi-asserted-by":"crossref","unstructured":"Chen, M.H., Kira, Z., AlRegib, G., Yoo, J., Chen, R., Zheng, J.: Temporal attentive alignment for large-scale video domain adaptation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6321\u20136330 (2019)","DOI":"10.1109\/ICCV.2019.00642"},{"key":"19_CR16","doi-asserted-by":"crossref","unstructured":"Chen, M.H., Li, B., Bao, Y., AlRegib, G.: Action segmentation with mixed temporal domain adaptation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 605\u2013614 (2020)","DOI":"10.1109\/WACV45572.2020.9093535"},{"key":"19_CR17","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1109\/TIP.2019.2928630","volume":"29","author":"Y Chen","year":"2019","unstructured":"Chen, Y., Song, S., Li, S., Wu, C.: A graph embedding framework for maximum mean discrepancy-based domain adaptation algorithms. IEEE Trans. Image Process. 29, 199\u2013213 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"19_CR18","doi-asserted-by":"crossref","unstructured":"Choi, J., Sharma, G., Chandraker, M., Huang, J.B.: Unsupervised and semi-supervised domain adaptation for action recognition from drones. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1717\u20131726 (2020)","DOI":"10.1109\/WACV45572.2020.9093511"},{"key":"19_CR19","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"678","DOI":"10.1007\/978-3-030-58610-2_40","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Choi","year":"2020","unstructured":"Choi, J., Sharma, G., Schulter, S., Huang, J.-B.: Shuffle and attend: video domain adaptation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12357, pp. 678\u2013695. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58610-2_40"},{"key":"19_CR20","doi-asserted-by":"crossref","unstructured":"Cormack, G.V., Clarke, C.L., Buettcher, S.: Reciprocal rank fusion outperforms condorcet and individual rank learning methods. In: International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 758\u2013759 (2009)","DOI":"10.1145\/1571941.1572114"},{"key":"19_CR21","doi-asserted-by":"crossref","unstructured":"Das, S., et al.: Toyota smarthome: real-world activities of daily living. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 833\u2013842 (2019)","DOI":"10.1109\/ICCV.2019.00092"},{"key":"19_CR22","doi-asserted-by":"crossref","unstructured":"Das, S., Dai, R., Yang, D., Bremond, F.: VPN++: rethinking video-pose embeddings for understanding activities of daily living. IEEE Trans. Pattern Anal. Mach. Intell. (2021)","DOI":"10.1109\/TPAMI.2021.3127885"},{"key":"19_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1007\/978-3-030-58545-7_5","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Das","year":"2020","unstructured":"Das, S., Sharma, S., Dai, R., Br\u00e9mond, F., Thonnat, M.: VPN: learning video-pose embedding for activities of daily living. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12354, pp. 72\u201390. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58545-7_5"},{"key":"19_CR24","doi-asserted-by":"crossref","unstructured":"Dawar, N., Kehtarnavaz, N.: A convolutional neural network-based sensor fusion system for monitoring transition movements in healthcare applications. In: 2018 IEEE 14th International Conference on Control and Automation (ICCA), pp. 482\u2013485. IEEE (2018)","DOI":"10.1109\/ICCA.2018.8444326"},{"issue":"1","key":"19_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/LSENS.2018.2878572","volume":"3","author":"N Dawar","year":"2018","unstructured":"Dawar, N., Ostadabbas, S., Kehtarnavaz, N.: Data augmentation in deep learning-based fusion of depth and inertial sensing for action recognition. IEEE Sens. Lett. 3(1), 1\u20134 (2018)","journal-title":"IEEE Sens. Lett."},{"key":"19_CR26","doi-asserted-by":"crossref","unstructured":"Delaitre, V., Laptev, I., Sivic, J.: Recognizing human actions in still images: a study of bag-of-features and part-based representations. In: BMVC 2010\u201321st British Machine Vision Conference (2010)","DOI":"10.5244\/C.24.97"},{"key":"19_CR27","doi-asserted-by":"publisher","first-page":"3835","DOI":"10.1109\/TIP.2020.2965299","volume":"29","author":"C Dhiman","year":"2020","unstructured":"Dhiman, C., Vishwakarma, D.K.: View-invariant deep architecture for human action recognition using two-stream motion and shape temporal dynamics. IEEE Trans. Image Process. 29, 3835\u20133844 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"19_CR28","doi-asserted-by":"crossref","unstructured":"Duan, H., Zhao, Y., Chen, K., Shao, D., Lin, D., Dai, B.: Revisiting skeleton-based action recognition. arXiv preprint arXiv:2104.13586 (2021)","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"19_CR29","doi-asserted-by":"crossref","unstructured":"Elekes, \u00c1., Sch\u00e4ler, M., B\u00f6hm, K.: On the various semantics of similarity in word embedding models. In: 2017 ACM\/IEEE Joint Conference on Digital Libraries (JCDL), pp. 1\u201310. IEEE (2017)","DOI":"10.1109\/JCDL.2017.7991568"},{"issue":"2","key":"19_CR30","doi-asserted-by":"publisher","first-page":"353","DOI":"10.1007\/s00355-011-0603-9","volume":"40","author":"P Emerson","year":"2013","unstructured":"Emerson, P.: The original borda count and partial voting. Social Choice Welfare 40(2), 353\u2013358 (2013)","journal-title":"Social Choice Welfare"},{"key":"19_CR31","doi-asserted-by":"publisher","unstructured":"van Erp, M., Vuurpijl, L., Schomaker, L.: An overview and comparison of voting methods for pattern recognition. In: Proceedings Eighth International Workshop on Frontiers in Handwriting Recognition, pp. 195\u2013200 (2002). https:\/\/doi.org\/10.1109\/IWFHR.2002.1030908","DOI":"10.1109\/IWFHR.2002.1030908"},{"key":"19_CR32","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Xie, S., Tai, Y.W., Lu, C.: RMPE: regional multi-person pose estimation. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.256"},{"key":"19_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"363","DOI":"10.1007\/3-540-45103-X_50","volume-title":"Image Analysis","author":"G Farneb\u00e4ck","year":"2003","unstructured":"Farneb\u00e4ck, G.: Two-frame motion estimation based on polynomial expansion. In: Bigun, J., Gustavsson, T. (eds.) SCIA 2003. LNCS, vol. 2749, pp. 363\u2013370. Springer, Heidelberg (2003). https:\/\/doi.org\/10.1007\/3-540-45103-X_50"},{"key":"19_CR34","doi-asserted-by":"crossref","unstructured":"Gao, R., Oh, T.H., Grauman, K., Torresani, L.: Listen to look: action recognition by previewing audio. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10457\u201310467 (2020)","DOI":"10.1109\/CVPR42600.2020.01047"},{"key":"19_CR35","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"898","DOI":"10.1007\/978-3-319-13560-1_76","volume-title":"PRICAI 2014: Trends in Artificial Intelligence","author":"M Ghifary","year":"2014","unstructured":"Ghifary, M., Kleijn, W.B., Zhang, M.: Domain adaptive neural networks for object recognition. In: Pham, D.-N., Park, S.-B. (eds.) PRICAI 2014. LNCS (LNAI), vol. 8862, pp. 898\u2013904. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-13560-1_76"},{"key":"19_CR36","doi-asserted-by":"crossref","unstructured":"Gretton, A., Borgwardt, K., Rasch, M., Sch\u00f6lkopf, B., Smola, A.: A kernel method for the two-sample-problem. Adv. Neural Inf. Process. Syst. 19 (2006)","DOI":"10.7551\/mitpress\/7503.003.0069"},{"issue":"1","key":"19_CR37","first-page":"723","volume":"13","author":"A Gretton","year":"2012","unstructured":"Gretton, A., Borgwardt, K.M., Rasch, M.J., Sch\u00f6lkopf, B., Smola, A.: A kernel two-sample test. J. Mach. Learn. Res. 13(1), 723\u2013773 (2012)","journal-title":"J. Mach. Learn. Res."},{"key":"19_CR38","unstructured":"Gretton, A., et al.: Optimal kernel choice for large-scale two-sample tests. Adv. Neural Inf. Process. Syst. 25 (2012)"},{"key":"19_CR39","unstructured":"Han, T., Xie, W., Zisserman, A.: Self-supervised co-training for video representation learning. In: NeurIPS (2020). http:\/\/arxiv.org\/abs\/2010.09709"},{"issue":"1","key":"19_CR40","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1109\/34.273716","volume":"16","author":"TK Ho","year":"1994","unstructured":"Ho, T.K., Hull, J.J., Srihari, S.N.: Decision combination in multiple classifier systems. IEEE Trans. Pattern Anal. Mach. Intell. 16(1), 66\u201375 (1994)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"19_CR41","doi-asserted-by":"publisher","unstructured":"Huber, P.J.: Robust estimation of a location parameter. In: Breakthroughs in Statistics, pp. 492\u2013518. Springer, Heidelberg (1992). https:\/\/doi.org\/10.1007\/978-1-4612-4380-9_35","DOI":"10.1007\/978-1-4612-4380-9_35"},{"key":"19_CR42","doi-asserted-by":"crossref","unstructured":"Imran, J., Kumar, P.: Human action recognition using rgb-d sensor and deep convolutional neural networks. In: 2016 International Conference on Advances in Computing, Communications and Informatics (ICACCI), pp. 144\u2013148. IEEE (2016)","DOI":"10.1109\/ICACCI.2016.7732038"},{"issue":"1","key":"19_CR43","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1007\/s12652-019-01239-9","volume":"11","author":"J Imran","year":"2020","unstructured":"Imran, J., Raman, B.: Evaluating fusion of rgb-d and inertial sensors for multimodal human action recognition. J. Ambient Intell. Hum. Comput. 11(1), 189\u2013208 (2020)","journal-title":"J. Ambient Intell. Hum. Comput."},{"key":"19_CR44","doi-asserted-by":"crossref","unstructured":"Jang, J., Kim, D., Park, C., Jang, M., Lee, J., Kim, J.: Etri-activity3d: a large-scale rgb-d dataset for robots to recognize daily activities of the elderly. In: 2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 10990\u201310997. IEEE (2020)","DOI":"10.1109\/IROS45743.2020.9341160"},{"issue":"9","key":"19_CR45","doi-asserted-by":"publisher","first-page":"1806","DOI":"10.1109\/TSMC.2018.2850149","volume":"49","author":"A Kamel","year":"2018","unstructured":"Kamel, A., Sheng, B., Yang, P., Li, P., Shen, R., Feng, D.D.: Deep convolutional neural networks for human action recognition using depth maps and postures. IEEE Trans. Syst. Man Cybern. Syst. 49(9), 1806\u20131819 (2018)","journal-title":"IEEE Trans. Syst. Man Cybern. Syst."},{"key":"19_CR46","doi-asserted-by":"crossref","unstructured":"Kampman, O., Barezi, E.J., Bertero, D., Fung, P.: Investigating audio, visual, and text fusion methods for end-to-end automatic personality prediction. arXiv preprint arXiv:1805.00705 (2018)","DOI":"10.18653\/v1\/P18-2096"},{"key":"19_CR47","doi-asserted-by":"crossref","unstructured":"Kazakos, E., Nagrani, A., Zisserman, A., Damen, D.: Epic-fusion: audio-visual temporal binding for egocentric action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5492\u20135501 (2019)","DOI":"10.1109\/ICCV.2019.00559"},{"key":"19_CR48","series-title":"Advances in Intelligent Systems and Computing","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1007\/978-981-10-7895-8_32","volume-title":"Proceedings of 2nd International Conference on Computer Vision & Image Processing","author":"P Khaire","year":"2018","unstructured":"Khaire, P., Imran, J., Kumar, P.: Human activity recognition by fusion of RGB, depth, and skeletal data. In: Chaudhuri, B.B., Kankanhalli, M.S., Raman, B. (eds.) Proceedings of 2nd International Conference on Computer Vision & Image Processing. AISC, vol. 703, pp. 409\u2013421. Springer, Singapore (2018). https:\/\/doi.org\/10.1007\/978-981-10-7895-8_32"},{"issue":"20","key":"19_CR49","doi-asserted-by":"publisher","first-page":"6774","DOI":"10.3390\/s21206774","volume":"21","author":"D Kim","year":"2021","unstructured":"Kim, D., Lee, I., Kim, D., Lee, S.: Action recognition using close-up of maximum activation and etri-activity3d livinglab dataset. Sensors 21(20), 6774 (2021)","journal-title":"Sensors"},{"key":"19_CR50","unstructured":"Korbar, B., Tran, D., Torresani, L.: Cooperative learning of audio and video models from self-supervised synchronization. In: Bengio, S., Wallach, H., Larochelle, H., Grauman, K., Cesa-Bianchi, N., Garnett, R. (eds.) Advances in Neural Information Processing Systems, vol. 31, pp. 7763\u20137774. Curran Associates, Inc. (2018)"},{"key":"19_CR51","doi-asserted-by":"crossref","unstructured":"Li, J., Wang, C., Zhu, H., Mao, Y., Fang, H.S., Lu, C.: Crowdpose: efficient crowded scenes pose estimation and a new benchmark. arXiv preprint arXiv:1812.00324 (2018)","DOI":"10.1109\/CVPR.2019.01112"},{"key":"19_CR52","unstructured":"Li, T., Wang, L.: Learning spatiotemporal features via video and text pair discrimination (2020)"},{"key":"19_CR53","doi-asserted-by":"crossref","unstructured":"Liang, T., Lin, G., Feng, L., Zhang, Y., Lv, F.: attention is not enough: mitigating the distribution discrepancy in asynchronous multimodal sequence fusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8148\u20138156 (2021)","DOI":"10.1109\/ICCV48922.2021.00804"},{"key":"19_CR54","doi-asserted-by":"crossref","unstructured":"Long, M., Wang, J., Ding, G., Sun, J., Yu, P.S.: Transfer feature learning with joint distribution adaptation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2200\u20132207 (2013)","DOI":"10.1109\/ICCV.2013.274"},{"key":"19_CR55","doi-asserted-by":"crossref","unstructured":"Martin, M., et al.: Drive &act: a multi-modal dataset for fine-grained driver behavior recognition in autonomous vehicles. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2801\u20132810 (2019)","DOI":"10.1109\/ICCV.2019.00289"},{"key":"19_CR56","doi-asserted-by":"crossref","unstructured":"Memmesheimer, R., Theisen, N., Paulus, D.: Gimme signals: discriminative signal encoding for multimodal activity recognition. In: 2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 10394\u201310401. IEEE (2020)","DOI":"10.1109\/IROS45743.2020.9341699"},{"key":"19_CR57","doi-asserted-by":"crossref","unstructured":"Munro, J., Damen, D.: Multi-modal domain adaptation for fine-grained action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 122\u2013132 (2020)","DOI":"10.1109\/CVPR42600.2020.00020"},{"issue":"2","key":"19_CR58","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1037\/1089-2680.2.2.175","volume":"2","author":"RS Nickerson","year":"1998","unstructured":"Nickerson, R.S.: Confirmation bias: a ubiquitous phenomenon in many guises. Rev. Gen. Psychol. 2(2), 175\u2013220 (1998)","journal-title":"Rev. Gen. Psychol."},{"key":"19_CR59","doi-asserted-by":"crossref","unstructured":"Pan, B., Cao, Z., Adeli, E., Niebles, J.C.: Adversarial cross-domain action recognition with co-attention. In: AAAI, vol. 34, pp. 11815\u201311822 (2020)","DOI":"10.1609\/aaai.v34i07.6854"},{"issue":"2","key":"19_CR60","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1109\/TNN.2010.2091281","volume":"22","author":"SJ Pan","year":"2010","unstructured":"Pan, S.J., Tsang, I.W., Kwok, J.T., Yang, Q.: Domain adaptation via transfer component analysis. IEEE Trans. Neural Netw. 22(2), 199\u2013210 (2010)","journal-title":"IEEE Trans. Neural Netw."},{"key":"19_CR61","doi-asserted-by":"crossref","unstructured":"Panda, R., et al.: AdaMML: adaptive multi-modal learning for efficient video recognition, pp. 7576\u20137585 (2021), https:\/\/openaccess.thecvf.com\/content\/ICCV2021\/html\/Panda_AdaMML_Adaptive_Multi-Modal_Learning_for_Efficient_Video_Recognition_ICCV_2021_paper.html","DOI":"10.1109\/ICCV48922.2021.00748"},{"key":"19_CR62","unstructured":"Patrick, M., Asano, Y., Fong, R., Henriques, J.F., Zweig, G., Vedaldi, A.: Multi-modal self-supervision from generalized data transformations. ArXiv abs\/2003.04298 (2020)"},{"issue":"19","key":"19_CR63","doi-asserted-by":"publisher","first-page":"28919","DOI":"10.1007\/s11042-021-11058-w","volume":"80","author":"C Pham","year":"2021","unstructured":"Pham, C., Nguyen, L., Nguyen, A., Nguyen, N., Nguyen, V.-T.: Combining skeleton and accelerometer data for human fine-grained activity recognition and abnormal behaviour detection with deep temporal convolutional networks. Multimedia Tools and Applications 80(19), 28919\u201328940 (2021). https:\/\/doi.org\/10.1007\/s11042-021-11058-w","journal-title":"Multimedia Tools and Applications"},{"key":"19_CR64","doi-asserted-by":"crossref","unstructured":"Piergiovanni, A., Angelova, A., Ryoo, M.S.: Evolving losses for unsupervised video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 133\u2013142 (2020)","DOI":"10.1109\/CVPR42600.2020.00021"},{"key":"19_CR65","doi-asserted-by":"crossref","unstructured":"Rai, N., Adeli, E., Lee, K.H., Gaidon, A., Niebles, J.C.: Cocon: cooperative-contrastive learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3384\u20133393 (2021)","DOI":"10.1109\/CVPRW53098.2021.00377"},{"issue":"1","key":"19_CR66","doi-asserted-by":"publisher","first-page":"44","DOI":"10.18178\/ijmlc.2019.9.1.763","volume":"9","author":"M Ramanathan","year":"2019","unstructured":"Ramanathan, M., Kochanowicz, J., Thalmann, N.M.: Combining pose-invariant kinematic features and object context features for RGB-D action recognition. Int. J. Mach. Learn. Comput. 9(1), 44\u201350 (2019)","journal-title":"Int. J. Mach. Learn. Comput."},{"key":"19_CR67","first-page":"3164","volume":"37","author":"SS Rani","year":"2021","unstructured":"Rani, S.S., Naidu, G.A., Shree, V.U.: Kinematic joint descriptor and depth motion descriptor with convolutional neural networks for human action recognition. Mater. Today: Proc. 37, 3164\u20133173 (2021)","journal-title":"Mater. Today: Proc."},{"key":"19_CR68","unstructured":"Redmon, J., Farhadi, A.: YOLOV3: an incremental improvement. arXiv preprint arXiv:1804.02767 (2018)"},{"key":"19_CR69","doi-asserted-by":"crossref","unstructured":"Rei\u00df, S., Roitberg, A., Haurilet, M., Stiefelhagen, R.: Deep classification-driven domain adaptation for cross-modal driver behavior recognition. In: 2020 IEEE Intelligent Vehicles Symposium (IV), pp. 1042\u20131047. IEEE (2020)","DOI":"10.1109\/IV47402.2020.9304782"},{"key":"19_CR70","doi-asserted-by":"crossref","unstructured":"Roitberg, A., Schneider, D., Djamal, A., Seibold, C., Rei\u00df, S., Stiefelhagen, R.: Let\u2019s play for action: recognizing activities of daily living by learning from life simulation video games. In: 2021 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 8563\u20138569. IEEE (2021)","DOI":"10.1109\/IROS51168.2021.9636381"},{"key":"19_CR71","doi-asserted-by":"crossref","unstructured":"Roitberg, A., Somani, N., Perzylo, A., Rickert, M., Knoll, A.: Multimodal human activity recognition for industrial manufacturing processes in robotic workcells. In: Proceedings of the 2015 ACM on International Conference on Multimodal Interaction, pp. 259\u2013266 (2015)","DOI":"10.1145\/2818346.2820738"},{"key":"19_CR72","doi-asserted-by":"crossref","unstructured":"Sankaranarayanan, S., Balaji, Y., Jain, A., Lim, S.N., Chellappa, R.: Learning from synthetic data: Addressing domain shift for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3752\u20133761 (2018)","DOI":"10.1109\/CVPR.2018.00395"},{"key":"19_CR73","doi-asserted-by":"crossref","unstructured":"Sharma, G., Jurie, F., Schmid, C.: Discriminative spatial saliency for image classification. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition, pp. 3506\u20133513. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6248093"},{"key":"19_CR74","doi-asserted-by":"crossref","unstructured":"Song, X., et al.: Spatio-temporal contrastive domain adaptation for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9787\u20139795 (2021)","DOI":"10.1109\/CVPR46437.2021.00966"},{"key":"19_CR75","unstructured":"Sun, Z., Ke, Q., Rahmani, H., Bennamoun, M., Wang, G., Liu, J.: Human action recognition from various data modalities: a review. arXiv preprint arXiv:2012.11866 (2020)"},{"key":"19_CR76","doi-asserted-by":"crossref","unstructured":"Wang, C., Yang, H., Meinel, C.: Exploring multimodal video representation for action recognition. In: International Joint Conference on Neural Networks, pp. 1924\u20131931. IEEE (2016)","DOI":"10.1109\/IJCNN.2016.7727435"},{"key":"19_CR77","doi-asserted-by":"crossref","unstructured":"Wang, L., Ding, Z., Tao, Z., Liu, Y., Fu, Y.: Generative multi-view human action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6212\u20136221 (2019)","DOI":"10.1109\/ICCV.2019.00631"},{"key":"19_CR78","doi-asserted-by":"crossref","unstructured":"Wang, P., Li, W., Gao, Z., Zhang, Y., Tang, C., Ogunbona, P.: Scene flow to action map: A new representation for rgb-d based action recognition with convolutional neural networks. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 595\u2013604 (2017)","DOI":"10.1109\/CVPR.2017.52"},{"key":"19_CR79","unstructured":"Wang, W., Li, H., Ding, Z., Wang, Z.: Rethink maximum mean discrepancy for domain adaptation. arXiv preprint arXiv:2007.00689 (2020)"},{"issue":"1","key":"19_CR80","doi-asserted-by":"publisher","first-page":"802","DOI":"10.1109\/TVCG.2021.3114794","volume":"28","author":"X Wang","year":"2021","unstructured":"Wang, X., He, J., Jin, Z., Yang, M., Wang, Y., Qu, H.: M2lens: visualizing and explaining multimodal models for sentiment analysis. IEEE Trans. Vis. Comput. Graph. 28(1), 802\u2013812 (2021)","journal-title":"IEEE Trans. Vis. Comput. Graph."},{"issue":"17","key":"19_CR81","doi-asserted-by":"publisher","first-page":"3680","DOI":"10.3390\/s19173680","volume":"19","author":"H Wei","year":"2019","unstructured":"Wei, H., Jafari, R., Kehtarnavaz, N.: Fusion of video and inertial sensing for deep learning-based human action recognition. Sensors 19(17), 3680 (2019)","journal-title":"Sensors"},{"issue":"3","key":"19_CR82","doi-asserted-by":"publisher","first-page":"254","DOI":"10.1037\/1082-989X.8.3.254","volume":"8","author":"RR Wilcox","year":"2003","unstructured":"Wilcox, R.R., Keselman, H.: Modern robust data analysis methods: measures of central tendency. Psychol. Methods 8(3), 254 (2003)","journal-title":"Psychol. Methods"},{"issue":"3","key":"19_CR83","doi-asserted-by":"publisher","first-page":"326","DOI":"10.1109\/TMM.2016.2520091","volume":"18","author":"P Wu","year":"2016","unstructured":"Wu, P., Liu, H., Li, X., Fan, T., Zhang, X.: A novel lip descriptor for audio-visual keyword spotting based on adaptive decision fusion. IEEE Trans. Multimedia 18(3), 326\u2013338 (2016)","journal-title":"IEEE Trans. Multimedia"},{"key":"19_CR84","unstructured":"Xiao, F., Lee, Y.J., Grauman, K., Malik, J., Feichtenhofer, C.: Audiovisual slowfast networks for video recognition. arXiv preprint arXiv:2001.08740 (2020)"},{"key":"19_CR85","doi-asserted-by":"crossref","unstructured":"Xie, S., Sun, C., Huang, J., Tu, Z., Murphy, K.: Rethinking spatiotemporal feature learning: speed-accuracy trade-offs in video classification. In: ECCV, pp. 305\u2013321 (2018)","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"19_CR86","unstructured":"Xiu, Y., Li, J., Wang, H., Fang, Y., Lu, C.: Pose flow: efficient online pose tracking. In: BMVC (2018)"},{"key":"19_CR87","doi-asserted-by":"crossref","unstructured":"Yan, H., Ding, Y., Li, P., Wang, Q., Xu, Y., Zuo, W.: Mind the class weight bias: weighted maximum mean discrepancy for unsupervised domain adaptation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2272\u20132281 (2017)","DOI":"10.1109\/CVPR.2017.107"},{"key":"19_CR88","doi-asserted-by":"publisher","first-page":"7989","DOI":"10.1109\/TPAMI.2021.3116945","volume":"44","author":"Z Yao","year":"2021","unstructured":"Yao, Z., Wang, Y., Wang, J., Yu, P., Long, M.: VIDEODG: generalizing temporal relations in videos to novel domains. IEEE Trans. Pattern Anal. Mach. Intell. 44, 7989\u20138004 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"19_CR89","doi-asserted-by":"crossref","unstructured":"Ye, J., Li, K., Qi, G.J., Hua, K.A.: Temporal order-preserving dynamic quantization for human action recognition from multimodal sensor streams. In: Proceedings of the 5th ACM on International Conference on Multimedia Retrieval, pp. 99\u2013106 (2015)","DOI":"10.1145\/2671188.2749340"},{"key":"19_CR90","unstructured":"Yi, C., Yang, S., Li, H., Tan, Y.P., Kot, A.: Benchmarking the robustness of spatial-temporal models against corruptions. arXiv preprint arXiv:2110.06513 (2021)"},{"issue":"4","key":"19_CR91","doi-asserted-by":"publisher","first-page":"716","DOI":"10.3390\/app9040716","volume":"9","author":"C Zhao","year":"2019","unstructured":"Zhao, C., Chen, M., Zhao, J., Wang, Q., Shen, Y.: 3D behavior recognition based on multi-modal deep space-time learning. Appl. Sci. 9(4), 716 (2019)","journal-title":"Appl. Sci."},{"issue":"1","key":"19_CR92","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1109\/TBDATA.2015.2465959","volume":"1","author":"Y Zheng","year":"2015","unstructured":"Zheng, Y.: Methodologies for cross-domain data fusion: an overview. IEEE Trans. Big Data 1(1), 16\u201334 (2015)","journal-title":"IEEE Trans. Big Data"},{"key":"19_CR93","doi-asserted-by":"crossref","unstructured":"Zou, H., Yang, J., Prasanna Das, H., Liu, H., Zhou, Y., Spanos, C.J.: Wifi and vision multimodal learning for accurate and robust device-free human activity recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (2019)","DOI":"10.1109\/CVPRW.2019.00056"},{"key":"19_CR94","doi-asserted-by":"publisher","first-page":"3197","DOI":"10.1109\/TIFS.2020.2985628","volume":"15","author":"Q Zou","year":"2020","unstructured":"Zou, Q., Wang, Y., Wang, Q., Zhao, Y., Li, Q.: Deep learning-based gait recognition using smartphones in the wild. IEEE Trans. Inf. Forensics Secur. 15, 3197\u20133212 (2020)","journal-title":"IEEE Trans. Inf. Forensics Secur."}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-25085-9_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T00:08:06Z","timestamp":1728864486000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-25085-9_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031250842","9783031250859"],"references-count":94,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-25085-9_19","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"12 February 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"From the workshops, 367 reviewed full papers have been selected for publication","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}