{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:05:02Z","timestamp":1777655102430,"version":"3.51.4"},"publisher-location":"Cham","reference-count":61,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030208721","type":"print"},{"value":"9783030208738","type":"electronic"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-20873-8_7","type":"book-chapter","created":{"date-parts":[[2019,5,25]],"date-time":"2019-05-25T16:32:03Z","timestamp":1558801923000},"page":"99-116","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["Cross Pixel Optical-Flow Similarity for Self-supervised Learning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2650-9871","authenticated-orcid":false,"given":"Aravindh","family":"Mahendran","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8410-2570","authenticated-orcid":false,"given":"James","family":"Thewlis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrea","family":"Vedaldi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,5,26]]},"reference":[{"key":"7_CR1","unstructured":"Abadi, M., et al.: TensorFlow: large-scale machine learning on heterogeneous distributed systems. arXiv preprint \n                      arXiv:1603.04467\n                      \n                     (2016)"},{"key":"7_CR2","doi-asserted-by":"crossref","unstructured":"Agrawal, P., Carreira, J., Malik, J.: Learning to see by moving. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.13"},{"key":"7_CR3","doi-asserted-by":"crossref","unstructured":"Arandjelovi\u0107, R., Zisserman, A.: Look, listen and learn. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"7_CR4","unstructured":"Bansal, A., Chen, X., Russell, B., Gupta, A., Ramanan, D.: PixelNet: representation of the pixels, by the pixels, and for the pixels. \n                      arXiv:1702.06506\n                      \n                     (2017)"},{"key":"7_CR5","unstructured":"Bojanowski, P., Joulin, A.: Unsupervised learning by predicting noise. In: ICML (2017)"},{"key":"7_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"611","DOI":"10.1007\/978-3-642-33783-3_44","volume-title":"Computer Vision \u2013 ECCV 2012","author":"DJ Butler","year":"2012","unstructured":"Butler, D.J., Wulff, J., Stanley, G.B., Black, M.J.: A naturalistic open source movie for optical flow evaluation. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012. LNCS, vol. 7577, pp. 611\u2013625. Springer, Heidelberg (2012). \n                      https:\/\/doi.org\/10.1007\/978-3-642-33783-3_44"},{"key":"7_CR7","volume-title":"An Introduction to Support Vector Machines","author":"N Cristianini","year":"2000","unstructured":"Cristianini, N., et al.: An Introduction to Support Vector Machines. CUP, Cambridge (2000)"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Doersch, C., Gupta, A., Efros, A.A.: Unsupervised visual representation learning by context prediction. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.167"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Doersch, C., et al.: Multi-task self-supervised visual learning. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.226"},{"key":"7_CR10","unstructured":"Donahue, J., et al.: Adversarial feature learning. In: ICLR (2017)"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., et al.: FlowNet: learning optical flow with convolutional networks. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.316"},{"issue":"9","key":"7_CR12","doi-asserted-by":"publisher","first-page":"1734","DOI":"10.1109\/TPAMI.2015.2496141","volume":"38","author":"A Dosovitskiy","year":"2016","unstructured":"Dosovitskiy, A., et al.: Discriminative unsupervised feature learning with exemplar convolutional neural networks. IEEE PAMI 38(9), 1734\u20131747 (2016)","journal-title":"IEEE PAMI"},{"key":"7_CR13","unstructured":"Everingham, M., et al.: The PASCAL visual object classes challenge 2007 results (2007)"},{"key":"7_CR14","unstructured":"Everingham, M., et al.: The PASCAL visual object classes challenge 2012 results (2012)"},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Faktor, A., Irani, M.: Video segmentation by non-local consensus voting. In: BMVC (2014)","DOI":"10.5244\/C.28.21"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Gan, C., Gong, B., Liu, K., Su, H., Guibas, L.J.: Geometry guided convolutional neural networks for self-supervised video representation learning. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00586"},{"key":"7_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"248","DOI":"10.1007\/978-3-319-54193-8_16","volume-title":"Computer Vision \u2013 ACCV 2016","author":"R Gao","year":"2017","unstructured":"Gao, R., Jayaraman, D., Grauman, K.: Object-centric representation learning from unlabeled videos. In: Lai, S.-H., Lepetit, V., Nishino, K., Sato, Y. (eds.) ACCV 2016. LNCS, vol. 10115, pp. 248\u2013263. Springer, Cham (2017). \n                      https:\/\/doi.org\/10.1007\/978-3-319-54193-8_16"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Urtasun, R.: Are we ready for autonomous driving? The KITTI vision benchmark suite. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"7_CR19","unstructured":"Gidaris, S., Singh, P., Komodakis, N.: Unsupervised representation learning by predicting image rotations. In: Proceedings of ICLR (2018)"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Girshick, R.B.: Fast R-CNN. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Hariharan, B., et al.: Semantic contours from inverse detectors. In: ICCV (2011)","DOI":"10.1109\/ICCV.2011.6126343"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Hariharan, B., Arbel\u00e1ez, P., Girshick, R., Malik, J.: Hypercolumns for object segmentation and fine-grained localization. In: CVPR, pp. 447\u2013456 (2015)","DOI":"10.1109\/CVPR.2015.7298642"},{"key":"7_CR23","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift. In: ICML (2015)"},{"key":"7_CR24","unstructured":"Isola, P., Zoran, D., Krishnan, D., Adelson, E.H.: Learning visual groups from co-occurrences in space and time. In: ICLR Workshop (2015)"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Jayaraman, D., Grauman, K.: Slow and steady feature analysis: higher order temporal coherence in video. In: CVPR, pp. 3852\u20133861 (2016)","DOI":"10.1109\/CVPR.2016.418"},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Jayaraman, D., et al.: Learning image representations tied to ego-motion. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.166"},{"key":"7_CR27","doi-asserted-by":"crossref","unstructured":"Jenni, S., Favaro, P.: Self-supervised feature learning by learning to spot artifacts. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00289"},{"key":"7_CR28","unstructured":"Kingma, D., Ba, J.: Adam: a method for stochastic optimization. \n                      arXiv:1412.6980\n                      \n                     (2014)"},{"key":"7_CR29","unstructured":"Kr\u00e4henb\u00fchl, P., et al.: Data-dependent initializations of convolutional neural networks. In: ICLR (2016)"},{"key":"7_CR30","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: ImageNet classification with deep convolutional neural networks. In: NIPS, pp. 1106\u20131114 (2012)"},{"key":"7_CR31","doi-asserted-by":"crossref","unstructured":"Larsson, G., Maire, M., Shakhnarovich, G.: Colorization as a proxy task for visual understanding. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.96"},{"key":"7_CR32","doi-asserted-by":"crossref","unstructured":"Lee, H.Y., Huang, J.B., Singh, M.K., Yang, M.H.: Unsupervised representation learning by sorting sequence. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.79"},{"key":"7_CR33","unstructured":"Liu, C.: Beyond pixels: exploring new representations and applications for motion analysis. Ph.D. thesis, Massachusetts Institute of Technology, USA (2009)"},{"key":"7_CR34","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-016-0911-8","volume":"120","author":"A Mahendran","year":"2016","unstructured":"Mahendran, A., Vedaldi, A.: Visualizing deep convolutional neural networks using natural pre-images. IJCV 120, 1\u201323 (2016)","journal-title":"IJCV"},{"key":"7_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"527","DOI":"10.1007\/978-3-319-46448-0_32","volume-title":"Computer Vision \u2013 ECCV 2016","author":"I Misra","year":"2016","unstructured":"Misra, I., Zitnick, C.L., Hebert, M.: Shuffle and learn: unsupervised learning using temporal order verification. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 527\u2013544. Springer, Cham (2016). \n                      https:\/\/doi.org\/10.1007\/978-3-319-46448-0_32"},{"key":"7_CR36","doi-asserted-by":"crossref","unstructured":"Mundhenk, T., Ho, D., Chen, B.Y.: Improvements to context based self-supervised learning. In: CVPR (2017)","DOI":"10.1109\/CVPR.2018.00973"},{"key":"7_CR37","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1007\/978-3-319-46466-4_5","volume-title":"Computer Vision \u2013 ECCV 2016","author":"M Noroozi","year":"2016","unstructured":"Noroozi, M., Favaro, P.: Unsupervised learning of visual representations by solving Jigsaw puzzles. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9910, pp. 69\u201384. Springer, Cham (2016). \n                      https:\/\/doi.org\/10.1007\/978-3-319-46466-4_5"},{"key":"7_CR38","doi-asserted-by":"crossref","unstructured":"Noroozi, M., Vinjimoor, A., Favaro, P., Pirsiavash, H.: Boosting self-supervised learning via knowledge transfer. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00975"},{"key":"7_CR39","doi-asserted-by":"crossref","unstructured":"Noroozi, M., et al.: Representation learning by learning to count. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.628"},{"key":"7_CR40","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"801","DOI":"10.1007\/978-3-319-46448-0_48","volume-title":"Computer Vision \u2013 ECCV 2016","author":"A Owens","year":"2016","unstructured":"Owens, A., Wu, J., McDermott, J.H., Freeman, W.T., Torralba, A.: Ambient sound provides supervision for visual learning. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 801\u2013816. Springer, Cham (2016). \n                      https:\/\/doi.org\/10.1007\/978-3-319-46448-0_48"},{"key":"7_CR41","doi-asserted-by":"crossref","unstructured":"Pathak, D., et al.: Context encoders: feature learning by inpainting. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.278"},{"key":"7_CR42","doi-asserted-by":"crossref","unstructured":"Pathak, D., et al.: Learning features by watching objects move. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.638"},{"key":"7_CR43","doi-asserted-by":"crossref","unstructured":"Prest, A., et al.: Learning object class detectors from weakly annotated video. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6248065"},{"key":"7_CR44","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: NIPS (2015)"},{"key":"7_CR45","doi-asserted-by":"crossref","unstructured":"Ren, Z., Lee, Y.J.: Cross-domain self-supervised multi-task feature learning using synthetic imagery. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00086"},{"key":"7_CR46","doi-asserted-by":"crossref","unstructured":"Revaud, J., Weinzaepfel, P., Harchaoui, Z., Schmid, C.: EpicFlow: edge-preserving interpolation of correspondences for optical flow. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7298720"},{"key":"7_CR47","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge. IJCV 115, 211\u2013252 (2015)","journal-title":"IJCV"},{"key":"7_CR48","unstructured":"de Sa, V.R.: Learning classification with unlabeled data. In: NIPS, pp. 112\u2013119 (1994)"},{"key":"7_CR49","doi-asserted-by":"crossref","unstructured":"Sermanet, P., et al.: Time-contrastive networks: self-supervised learning from video. In: Proceedings of International Conference on Robotics and Automation (2018)","DOI":"10.1109\/ICRA.2018.8462891"},{"key":"7_CR50","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. \n                      arXiv:1409.1556\n                      \n                     (2014)"},{"key":"7_CR51","unstructured":"Srivastava, N., Mansimov, E., Salakhudinov, R.: Unsupervised learning of video representations using LSTMs. In: ICML (2015)"},{"key":"7_CR52","doi-asserted-by":"crossref","unstructured":"Thomee, B., et\u00a0al.: YFCC100M: the new data in multimedia research. ACM (2016)","DOI":"10.1145\/2812802"},{"issue":"12","key":"7_CR53","doi-asserted-by":"publisher","first-page":"5345","DOI":"10.4249\/scholarpedia.5345","volume":"3","author":"D Todorovic","year":"2008","unstructured":"Todorovic, D.: Gestalt principles. Scholarpedia 3(12), 5345 (2008). revision #91314","journal-title":"Scholarpedia"},{"key":"7_CR54","unstructured":"Walker, J.: Data-driven visual forecasting. Ph.D. thesis, Carnegie Mellon University (2018)"},{"key":"7_CR55","doi-asserted-by":"crossref","unstructured":"Wang, X., Gupta, A.: Unsupervised learning of visual representations using videos. In: ICCV, pp. 2794\u20132802 (2015)","DOI":"10.1109\/ICCV.2015.320"},{"key":"7_CR56","doi-asserted-by":"crossref","unstructured":"Wang, X., He, K., Gupta, A.: Transitive invariance for self-supervised visual representation learning. In: ICCV, pp. 2794\u20132802 (2017)","DOI":"10.1109\/ICCV.2017.149"},{"key":"7_CR57","doi-asserted-by":"crossref","unstructured":"Wei, D., et al.: Learning and using the arrow of time. In: CVPR, pp. 8052\u20138060 (2018)","DOI":"10.1109\/CVPR.2018.00840"},{"key":"7_CR58","doi-asserted-by":"crossref","unstructured":"Weinzaepfel, P., Revaud, J., Harchaoui, Z., Schmid, C.: DeepFlow: large displacement optical flow with deep matching. In: ICCV, pp. 1385\u20131392 (2013)","DOI":"10.1109\/ICCV.2013.175"},{"key":"7_CR59","doi-asserted-by":"publisher","unstructured":"Xue, T., Wu, J., Bouman, K.L., Freeman, W.T.: Visual dynamics: stochastic future generation via layered cross convolutional networks. IEEE PAMI (2018). \n                      https:\/\/ieeexplore.ieee.org\/document\/8409321\n                      \n                    . \n                      https:\/\/doi.org\/10.1109\/TPAMI.2018.2854726","DOI":"10.1109\/TPAMI.2018.2854726"},{"key":"7_CR60","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"649","DOI":"10.1007\/978-3-319-46487-9_40","volume-title":"Computer Vision \u2013 ECCV 2016","author":"R Zhang","year":"2016","unstructured":"Zhang, R., Isola, P., Efros, A.A.: Colorful image colorization. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9907, pp. 649\u2013666. Springer, Cham (2016). \n                      https:\/\/doi.org\/10.1007\/978-3-319-46487-9_40"},{"key":"7_CR61","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A.: Split-brain autoencoders: unsupervised learning by cross-channel prediction. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.76"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ACCV 2018"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-20873-8_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,25]],"date-time":"2019-05-25T16:33:41Z","timestamp":1558802021000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-20873-8_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030208721","9783030208738"],"references-count":61,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-20873-8_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"26 May 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ACCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Perth, WA","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 December 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"accv2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/accv2018.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"Microsoft CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"979","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"274","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"2.7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}}]}}