{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T14:51:19Z","timestamp":1786114279112,"version":"3.56.0"},"publisher-location":"Cham","reference-count":79,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031197772","type":"print"},{"value":"9783031197789","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-19778-9_38","type":"book-chapter","created":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T20:28:41Z","timestamp":1667420921000},"page":"657-675","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":51,"title":["My View is the Best View: Procedure Learning from\u00a0Egocentric Videos"],"prefix":"10.1007","author":[{"given":"Siddhant","family":"Bansal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chetan","family":"Arora","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"C. V.","family":"Jawahar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,3]]},"reference":[{"key":"38_CR1","unstructured":"Ahsan, U., Sun, C., Essa, I.: DiscrimNet: semi-supervised action recognition from videos using generative adversarial networks. In: Computer Vision and Pattern Recognition Workshops (CVPRW) \u2018Women in Computer Vision (WiCV)\u2019 (2018)"},{"key":"38_CR2","doi-asserted-by":"crossref","unstructured":"Alayrac, J.B., Bojanowski, P., Agrawal, N., Laptev, I., Sivic, J., Lacoste-Julien, S.: Unsupervised learning from narrated instruction videos. In: Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.495"},{"key":"38_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"628","DOI":"10.1007\/978-3-319-10602-1_41","volume-title":"Computer Vision \u2013 ECCV 2014","author":"P Bojanowski","year":"2014","unstructured":"Bojanowski, P., et al.: Weakly supervised action labeling in videos under ordering constraints. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 628\u2013643. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_41"},{"key":"38_CR4","doi-asserted-by":"crossref","unstructured":"Boykov, Y., Veksler, O., Zabih, R.: Fast approximate energy minimization via graph cuts. IEEE Trans. Pattern Anal. Mach. Intell. (2001)","DOI":"10.1109\/34.969114"},{"key":"38_CR5","doi-asserted-by":"crossref","unstructured":"Carlucci, F.M., D\u2019Innocente, A., Bucci, S., Caputo, B., Tommasi, T.: Domain generalization by solving Jigsaw puzzles. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00233"},{"key":"38_CR6","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo Vadis, action recognition? A new model and the kinetics dataset. In: Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"38_CR7","doi-asserted-by":"crossref","unstructured":"Chang, C.Y., Huang, D.A., Sui, Y., Fei-Fei, L., Niebles, J.C.: D3TW: discriminative differentiable dynamic time warping for weakly supervised action alignment and segmentation. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00366"},{"key":"38_CR8","doi-asserted-by":"crossref","unstructured":"Conners, R.W., Harlow, C.A.: A theoretical comparison of texture algorithms. IEEE Trans. Pattern Anal. Mach. Intell. (1980)","DOI":"10.1109\/TPAMI.1980.4767008"},{"key":"38_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"753","DOI":"10.1007\/978-3-030-01225-0_44","volume-title":"Computer Vision \u2013 ECCV 2018","author":"D Damen","year":"2018","unstructured":"Damen, D., et al.: Scaling egocentric vision: the EPIC-KITCHENS dataset. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11208, pp. 753\u2013771. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01225-0_44"},{"key":"38_CR10","doi-asserted-by":"crossref","unstructured":"Damen, D., Leelasawassuk, T., Haines, O., Calway, A., Mayol-Cuevas, W.: You-Do, I-Learn: discovering task relevant objects and their modes of interaction from multi-user egocentric video. In: British Machine Vision Conference (BMVC) (2014)","DOI":"10.5244\/C.28.30"},{"key":"38_CR11","unstructured":"De La Torre, F., et al.: Guide to the Carnegie Mellon University Multimodal Activity (CMU-MMAC) database. In: Robotics Institute (2008)"},{"key":"38_CR12","doi-asserted-by":"crossref","unstructured":"Diba, A., Sharma, V., Gool, L., Stiefelhagen, R.: DynamoNet: dynamic action and motion network. In: International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00629"},{"key":"38_CR13","unstructured":"Ding, L., Xu, C.: Weakly-supervised action segmentation with iterative soft boundary assignment. In: Computer Vision and Pattern Recognition (CVPR) (2018)"},{"key":"38_CR14","doi-asserted-by":"crossref","unstructured":"Doughty, H., Laptev, I., Mayol-Cuevas, W., Damen, D.: Action modifiers: learning from adverbs in instructional videos. In: Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00095"},{"key":"38_CR15","doi-asserted-by":"crossref","unstructured":"Dunn, J.C.: A fuzzy relative of the ISODATA process and its use in detecting compact well-separated clusters. J. Cybern. (1973)","DOI":"10.1080\/01969727308546046"},{"key":"38_CR16","doi-asserted-by":"crossref","unstructured":"Dwibedi, D., Aytar, Y., Tompson, J., Sermanet, P., Zisserman, A.: Temporal cycle-consistency learning. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00190"},{"key":"38_CR17","unstructured":"ELAN (Version 6.0) [Computer software] (2020). Nijmegen: Max Planck Institute for Psycholinguistics, The Language Archive: https:\/\/archive.mpi.nl\/tla\/elan"},{"key":"38_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"557","DOI":"10.1007\/978-3-030-58520-4_33","volume-title":"Computer Vision \u2013 ECCV 2020","author":"E Elhamifar","year":"2020","unstructured":"Elhamifar, E., Huynh, D.: Self-supervised multi-task procedure learning from instructional videos. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12362, pp. 557\u2013573. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58520-4_33"},{"key":"38_CR19","doi-asserted-by":"crossref","unstructured":"Elhamifar, E., Naing, Z.: Unsupervised procedure learning via joint dynamic summarization. In: International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00644"},{"key":"38_CR20","doi-asserted-by":"crossref","unstructured":"Feng, Z., Xu, C., Tao, D.: Self-supervised representation learning by rotation feature decoupling. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.01061"},{"key":"38_CR21","doi-asserted-by":"crossref","unstructured":"Fernando, B., Bilen, H., Gavves, E., Gould, S.: Self-supervised video representation learning with odd-one-out networks. In: Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.607"},{"key":"38_CR22","doi-asserted-by":"crossref","unstructured":"Fried, D., Alayrac, J.B., Blunsom, P., Dyer, C., Clark, S., Nematzadeh, A.: Learning to segment actions from observation and narration. In: Association for Computational Linguistics (ACL) (2020)","DOI":"10.18653\/v1\/2020.acl-main.231"},{"key":"38_CR23","doi-asserted-by":"crossref","unstructured":"Furnari, A., Farinella, G.: Rolling-unrolling LSTMs for action anticipation from first-person video. IEEE Trans. Pattern Anal. Mach. Intell. (2020)","DOI":"10.1109\/TPAMI.2020.2992889"},{"key":"38_CR24","unstructured":"Grauman, K., et al.: Ego4D: around the world in 3,000 hours of egocentric video. In: Computer Vision and Pattern Recognition (CVPR) (2022)"},{"key":"38_CR25","doi-asserted-by":"crossref","unstructured":"Greig, D., Porteous, B., Seheult, A.: Exact maximum a posteriori estimation for binary images. J. Roy. Stat. Soc. Ser. B-Methodol. (1989)","DOI":"10.1111\/j.2517-6161.1989.tb01764.x"},{"key":"38_CR26","doi-asserted-by":"crossref","unstructured":"Han, T., Xie, W., Zisserman, A.: Video representation learning by dense predictive coding. In: Workshop on Large Scale Holistic Video Understanding, ICCV (2019)","DOI":"10.1109\/ICCVW.2019.00186"},{"key":"38_CR27","doi-asserted-by":"crossref","unstructured":"Haresh, S., et al.: Learning by aligning videos in time. In: Computer Vision and Pattern Recognition (CVPR) (2021)","DOI":"10.1109\/CVPR46437.2021.00550"},{"key":"38_CR28","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"38_CR29","unstructured":"Hinton, G.E., Zemel, R.S.: Autoencoders, minimum description length and helmholtz free energy. In: Neural Information Processing Systems (1993)"},{"key":"38_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/978-3-319-46493-0_9","volume-title":"Computer Vision \u2013 ECCV 2016","author":"D-A Huang","year":"2016","unstructured":"Huang, D.-A., Fei-Fei, L., Niebles, J.C.: Connectionist temporal modeling for weakly supervised action labeling. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 137\u2013153. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_9"},{"key":"38_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"789","DOI":"10.1007\/978-3-030-01225-0_46","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Y Huang","year":"2018","unstructured":"Huang, Y., Cai, M., Li, Z., Sato, Y.: Predicting gaze in egocentric video by learning task-dependent attention transition. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11208, pp. 789\u2013804. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01225-0_46"},{"key":"38_CR32","doi-asserted-by":"crossref","unstructured":"Jang, Y., Sullivan, B., Ludwig, C., Gilchrist, I., Damen, D., Mayol-Cuevas, W.: EPIC-tent: an egocentric video dataset for camping tent assembly. In: International Conference on Computer Vision (ICCV) Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00547"},{"key":"38_CR33","doi-asserted-by":"crossref","unstructured":"Ji, L., et al.: Learning temporal video procedure segmentation from an automatically collected large dataset. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV) (2022)","DOI":"10.1109\/WACV51458.2022.00279"},{"key":"38_CR34","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"678","DOI":"10.1007\/978-3-030-58610-2_40","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Choi","year":"2020","unstructured":"Choi, J., Sharma, G., Schulter, S., Huang, J.-B.: Shuffle and attend: video domain adaptation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12357, pp. 678\u2013695. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58610-2_40"},{"key":"38_CR35","doi-asserted-by":"crossref","unstructured":"Kim, D., Cho, D., Kweon, I.S.: Self-supervised video representation learning with space-time cubic puzzles. In: AAAI Conference on Artificial Intelligence (2019)","DOI":"10.1609\/aaai.v33i01.33018545"},{"key":"38_CR36","doi-asserted-by":"crossref","unstructured":"Kim, D., Cho, D., Yoo, D., Kweon, I.S.: Learning image representations by completing damaged Jigsaw puzzles. In: Winter Conference on Applications of Computer Vision (WACV) (2018)","DOI":"10.1109\/WACV.2018.00092"},{"key":"38_CR37","unstructured":"Komodakis, N., Gidaris, S.: Unsupervised representation learning by predicting image rotations. In: International Conference on Learning Representations (ICLR) (2018)"},{"key":"38_CR38","unstructured":"Kuehne, H., Arslan, A.B., Serre, T.: The language of actions: recovering the syntax and semantics of goal-directed human activities. In: Computer Vision and Pattern Recognition (CVPR) (2016)"},{"key":"38_CR39","doi-asserted-by":"crossref","unstructured":"Kuhn, H.W.: The Hungarian method for the assignment problem. Naval Res. Logist. Q. (1955)","DOI":"10.1002\/nav.3800020109"},{"key":"38_CR40","doi-asserted-by":"crossref","unstructured":"Kukleva, A., Kuehne, H., Sener, F., Gall, J.: Unsupervised learning of action classes with continuous temporal embedding. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.01234"},{"key":"38_CR41","doi-asserted-by":"crossref","unstructured":"Larsson, G., Maire, M., Shakhnarovich, G.: Colorization as a proxy task for visual understanding. In: Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.96"},{"key":"38_CR42","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"577","DOI":"10.1007\/978-3-319-46493-0_35","volume-title":"Computer Vision \u2013 ECCV 2016","author":"G Larsson","year":"2016","unstructured":"Larsson, G., Maire, M., Shakhnarovich, G.: Learning representations for automatic colorization. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 577\u2013593. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_35"},{"key":"38_CR43","doi-asserted-by":"crossref","unstructured":"Lee, H.Y., Huang, J.B., Singh, M.K., Yang, M.H.: Unsupervised representation learning by sorting sequences. In: International Conference on Computer Vision (ICCV) (2017)","DOI":"10.1109\/ICCV.2017.79"},{"key":"38_CR44","doi-asserted-by":"crossref","unstructured":"Li, J., Lei, P., Todorovic, S.: Weakly supervised energy-based learning for action segmentation. In: International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00634"},{"key":"38_CR45","doi-asserted-by":"crossref","unstructured":"Li, J., Todorovic, S.: Set-constrained viterbi for set-supervised action segmentation. In: Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01083"},{"key":"38_CR46","doi-asserted-by":"crossref","unstructured":"Li, Y., Fathi, A., Rehg, J.M.: Learning to predict gaze in egocentric video. In: International Conference on Computer Vision (ICCV) (2013)","DOI":"10.1109\/ICCV.2013.399"},{"key":"38_CR47","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"639","DOI":"10.1007\/978-3-030-01228-1_38","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Y Li","year":"2018","unstructured":"Li, Y., Liu, M., Rehg, J.M.: In the eye of beholder: joint learning of gaze and actions in first person video. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11209, pp. 639\u2013655. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01228-1_38"},{"key":"38_CR48","doi-asserted-by":"crossref","unstructured":"Liu, X., van de Weijer, J., Bagdanov, A.D.: Leveraging unlabeled data for crowd counting by learning to rank. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00799"},{"key":"38_CR49","doi-asserted-by":"crossref","unstructured":"Lloyd, S.: Least squares quantization in PCM. IEEE Trans. Inf. Theory (1982)","DOI":"10.1109\/TIT.1982.1056489"},{"key":"38_CR50","doi-asserted-by":"crossref","unstructured":"Malmaud, J., Huang, J., Rathod, V., Johnston, N., Rabinovich, A., Murphy, K.: What\u2019s Cookin\u2019? Interpreting cooking videos using text. speech and vision. In: HLT-NAACL (2015)","DOI":"10.3115\/v1\/N15-1015"},{"key":"38_CR51","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.B., Tapaswi, M., Laptev, I., Sivic, J.: HowTo100M: learning a text-video embedding by watching hundred million narrated video clips. In: International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00272"},{"key":"38_CR52","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"527","DOI":"10.1007\/978-3-319-46448-0_32","volume-title":"Computer Vision \u2013 ECCV 2016","author":"I Misra","year":"2016","unstructured":"Misra, I., Zitnick, C.L., Hebert, M.: Shuffle and learn: unsupervised learning using temporal order verification. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 527\u2013544. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_32"},{"key":"38_CR53","unstructured":"Naing, Z., Elhamifar, E.: Procedure completion by learning from partial summaries. In: British Machine Vision Conference (BMVC) (2020)"},{"key":"38_CR54","doi-asserted-by":"crossref","unstructured":"Ng, E., Xiang, D., Joo, H., Grauman, K.: You2Me: inferring body pose in egocentric video via first and second person interactions. In: Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00991"},{"key":"38_CR55","doi-asserted-by":"crossref","unstructured":"Noroozi, M., Pirsiavash, H., Favaro, P.: Representation learning by learning to count. In: International Conference on Computer Vision (ICCV) (2017)","DOI":"10.1109\/ICCV.2017.628"},{"key":"38_CR56","unstructured":"Paszke, A., et al.: PyTorch: an imperative style, high-performance deep learning library. In: Neural Information Processing Systems (2019)"},{"key":"38_CR57","doi-asserted-by":"crossref","unstructured":"Pirsiavash, H., Ramanan, D.: Detecting activities of daily living in first-person camera views. In: Computer Vision and Pattern Recognition (CVPR) (2012)","DOI":"10.1109\/CVPR.2012.6248010"},{"key":"38_CR58","doi-asserted-by":"crossref","unstructured":"Ragusa, F., Furnari, A., Livatino, S., Farinella, G.M.: The MECCANO dataset: understanding human-object interactions from egocentric videos in an industrial-like domain. In: Winter Conference on Applications of Computer Vision (WACV), pp. 1569\u20131578 (2021)","DOI":"10.1109\/WACV48630.2021.00161"},{"key":"38_CR59","doi-asserted-by":"crossref","unstructured":"Richard, A., Kuehne, H., Gall, J.: Action sets: weakly supervised action segmentation without ordering constraints. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00627"},{"key":"38_CR60","doi-asserted-by":"crossref","unstructured":"Richard, A., Kuehne, H., Iqbal, A., Gall, J.: NeuralNetwork-viterbi: a framework for weakly supervised video learning. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00771"},{"key":"38_CR61","doi-asserted-by":"crossref","unstructured":"Sener, F., Yao, A.: Zero-shot anticipation for instructional activities. In: International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00095"},{"key":"38_CR62","doi-asserted-by":"crossref","unstructured":"Sener, O., Zamir, A.R., Savarese, S., Saxena, A.: Unsupervised semantic parsing of video collections. In: International Conference on Computer Vision (ICCV) (2015)","DOI":"10.1109\/ICCV.2015.509"},{"key":"38_CR63","doi-asserted-by":"crossref","unstructured":"Shen, Y., Wang, L., Elhamifar, E.: Learning To segment actions from visual and language instructions via differentiable weak sequence alignment. In: Computer Vision and Pattern Recognition (CVPR) (2021)","DOI":"10.1109\/CVPR46437.2021.01002"},{"key":"38_CR64","doi-asserted-by":"crossref","unstructured":"Sigurdsson, G.A., Gupta, A., Schmid, C., Farhadi, A., Alahari, K.: Actor and observer: joint modeling of first and third-person videos. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00772"},{"key":"38_CR65","doi-asserted-by":"crossref","unstructured":"Singh, S., Arora, C., Jawahar, C.V.: First person action recognition using deep learned descriptors. In: Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.287"},{"key":"38_CR66","unstructured":"Srivastava, N., Mansimov, E., Salakhutdinov, R.: Unsupervised learning of video representations using LSTMs. In: International Conference on Machine Learning (ICML) (2015)"},{"key":"38_CR67","doi-asserted-by":"crossref","unstructured":"Tang, Y., et al.: COIN: a large-scale dataset for comprehensive instructional video analysis. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00130"},{"key":"38_CR68","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L.D., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: International Conference on Computer Vision (ICCV) (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"38_CR69","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., Paluri, M.: A closer look at spatiotemporal convolutions for action recognition. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00675"},{"key":"38_CR70","doi-asserted-by":"crossref","unstructured":"VidalMata, R.G., Scheirer, W.J., Kukleva, A., Cox, D., Kuehne, H.: Joint visual-temporal embedding for unsupervised learning of actions in untrimmed sequences. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV) (2021)","DOI":"10.1109\/WACV48630.2021.00128"},{"key":"38_CR71","doi-asserted-by":"crossref","unstructured":"Vincent, P., Larochelle, H., Bengio, Y., Manzagol, P.A.: Extracting and composing robust features with denoising autoencoders. In: International Conference on Machine Learning (ICML) (2008)","DOI":"10.1145\/1390156.1390294"},{"key":"38_CR72","unstructured":"Vondrick, C., Pirsiavash, H., Torralba, A.: Generating videos with scene dynamics. In: Neural Information Processing Systems (2016)"},{"key":"38_CR73","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R.B., Gupta, A., He, K.: Non-local neural networks. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"38_CR74","doi-asserted-by":"crossref","unstructured":"Wei, D., Lim, o., Zisserman, A., Freeman, W.T.: Learning and using the arrow of time. In: Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00840"},{"key":"38_CR75","doi-asserted-by":"crossref","unstructured":"Xu, D., Xiao, J., Zhao, Z., Shao, J., Xie, D., Zhuang, Y.: Self-supervised spatiotemporal learning via video clip order prediction. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.01058"},{"key":"38_CR76","doi-asserted-by":"crossref","unstructured":"Yu, S.I., Jiang, L., Hauptmann, A.: Instructional videos for unsupervised harvesting and learning of action examples. In: ACM International Conference on Multimedia (2014)","DOI":"10.1145\/2647868.2654997"},{"key":"38_CR77","doi-asserted-by":"crossref","unstructured":"Zhou, L., Xu, C., Corso, J.J.: Towards automatic learning of procedures from web instructional videos. In: AAAI Conference on Artificial Intelligence (2018)","DOI":"10.1609\/aaai.v32i1.12342"},{"key":"38_CR78","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"470","DOI":"10.1007\/978-3-030-58526-6_28","volume-title":"Computer Vision \u2013 ECCV 2020","author":"D Zhukov","year":"2020","unstructured":"Zhukov, D., Alayrac, J.-B., Laptev, I., Sivic, J.: Learning actionness via long-range temporal order verification. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12374, pp. 470\u2013487. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58526-6_28"},{"key":"38_CR79","doi-asserted-by":"crossref","unstructured":"Zhukov, D., Alayrac, J.B., Cinbis, R.G., Fouhey, D., Laptev, I., Sivic, J.: Cross-task weakly supervised learning from instructional videos. In: Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00365"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-19778-9_38","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T21:03:07Z","timestamp":1667422987000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-19778-9_38"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031197772","9783031197789"],"references-count":79,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-19778-9_38","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"3 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}