{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T11:51:26Z","timestamp":1783425086821,"version":"3.54.6"},"reference-count":207,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Pioneer Centre for AI"},{"DOI":"10.13039\/501100001732","name":"Danmarks Grundforskningsfond","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001732","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Spanish project","award":["PID2019-105093GB-I00"],"award-info":[{"award-number":["PID2019-105093GB-I00"]}]},{"DOI":"10.13039\/501100003741","name":"Instituci\u00f3 Catalana de Recerca i Estudis Avan\u00e7ats","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003741","id-type":"DOI","asserted-by":"publisher"}]},{"name":"ICREA Academia program"},{"DOI":"10.13039\/501100002702","name":"Aalborg Universitet","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002702","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2023,11,1]]},"DOI":"10.1109\/tpami.2023.3243465","type":"journal-article","created":{"date-parts":[[2023,2,9]],"date-time":"2023-02-09T18:37:51Z","timestamp":1675967871000},"page":"12922-12943","source":"Crossref","is-referenced-by-count":138,"title":["Video Transformers: A Survey"],"prefix":"10.1109","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6316-4190","authenticated-orcid":false,"given":"Javier","family":"Selva","sequence":"first","affiliation":[{"name":"Computer Vision Center, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9330-522X","authenticated-orcid":false,"given":"Anders S.","family":"Johansen","sequence":"additional","affiliation":[{"name":"Milestone Systems, Br&#x00F8;ndby, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0617-8873","authenticated-orcid":false,"given":"Sergio","family":"Escalera","sequence":"additional","affiliation":[{"name":"Computer Vision Center, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1953-0429","authenticated-orcid":false,"given":"Kamal","family":"Nasrollahi","sequence":"additional","affiliation":[{"name":"Milestone Systems, Br&#x00F8;ndby, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7584-5209","authenticated-orcid":false,"given":"Thomas B.","family":"Moeslund","sequence":"additional","affiliation":[{"name":"Pioneer Center for AI, Copenhagen, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4089-9060","authenticated-orcid":false,"given":"Albert","family":"Clap\u00e9s","sequence":"additional","affiliation":[{"name":"Computer Vision Center, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2021.104216"},{"key":"ref207","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"ref56","first-page":"802","article-title":"Convolutional LSTM network: A machine learning approach for precipitation nowcasting","author":"shi","year":"2015","journal-title":"Proc 28th Int Conf Neural Inf Process Syst"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00808"},{"key":"ref205","first-page":"12310","article-title":"Barlow twins: Self-supervised learning via redundancy reduction","author":"zbontar","year":"2021","journal-title":"Proc 38th Int Conf Mach Learn"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00256"},{"key":"ref206","first-page":"5228","article-title":"VICReg: Variance-invariance-covariance regularization for self-supervised learning","author":"bardes","year":"2022","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref53","article-title":"A short note on the kinetics-700 human action dataset","author":"carreira","year":"2019"},{"key":"ref203","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00517"},{"key":"ref52","first-page":"318","article-title":"Rethinking spatiotemporal feature learning for video understanding","author":"xie","year":"2017","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref204","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018545"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2019.8803639"},{"key":"ref201","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1002"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00272"},{"key":"ref202","first-page":"23296","article-title":"Intriguing properties of vision transformers","author":"naseer","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref51","article-title":"Contrastive bidirectional transformer for temporal representation learning","author":"sun","year":"2019"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58598-3_27"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1080\/21681163.2020.1835550"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6718"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3065823"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00481"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.233"},{"key":"ref41","article-title":"ImageNet-21k pretraining for the masses","author":"ridnik","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref44","article-title":"Deepfake detection scheme based on vision transformer and distillation","author":"heo","year":"2021"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00125"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01378"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00033"},{"key":"ref7","first-page":"813","article-title":"Is space-time attention all you need for video understanding?","author":"bertasius","year":"2021","journal-title":"Proc 38th Int Conf Mach Learn"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-698"},{"key":"ref3","article-title":"Skeleton-based action recognition via spatial and temporal transformer networks","volume":"208","author":"plizzari","year":"2021","journal-title":"Comput Vis Image Understanding"},{"key":"ref6","first-page":"213","article-title":"End-to-end object detection with transformers","author":"carion","year":"2020","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref5","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"dosovitskiy","year":"2021","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01940"},{"key":"ref101","first-page":"25804","article-title":"Time is MattEr: Temporal self-supervision for video transformers","author":"yun","year":"2022","journal-title":"Proc 39th Int Conf Mach Learn"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref35","article-title":"COOT: Cooperative hierarchical transformer for video-text representation learning","author":"ging","year":"2020","journal-title":"Proc 34th Int Conf Neural Inf Process Syst"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01248"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-68238-5_48"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00747"},{"key":"ref31","article-title":"Improving language understanding by generative pre-training","author":"radford","year":"2018","journal-title":"OpenAI Preprint"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01327"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00970"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.634"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2022.01.001"},{"key":"ref23","article-title":"Multimodal learning with transformers: A survey","author":"xu","year":"2022"},{"key":"ref26","article-title":"On the effectiveness of task granularity for transfer learning","author":"mahdisoltani","year":"2018"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01547-8"},{"key":"ref21","article-title":"Transformers meet visual learning understanding: A comprehensive review","author":"yang","year":"2022"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00813"},{"key":"ref29","article-title":"VideoMAE: Masked autoencoders are data-efficient learners for self-supervised video pre-training","author":"tong","year":"2022","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref200","first-page":"3543","article-title":"Attention is not explanation","author":"jain","year":"2019","journal-title":"Proc Conf North Amer Chapter Assoc Comput Linguistics Hum Lang Technol"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00333"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"ref97","first-page":"19594","article-title":"Space-time mixing attention for video transformer","author":"bulat","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00290"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00015"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00585"},{"key":"ref124","first-page":"5099","article-title":"PointNet: Deep hierarchical feature learning on point sets in a metric space","author":"qi","year":"2017","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref98","article-title":"Longformer: The long-document transformer","author":"beltagy","year":"2020"},{"key":"ref125","first-page":"26462","article-title":"Learning from inside: Self-driven siamese sampling and reasoning for video question answering","author":"yu","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01606"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00610"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00355"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"ref95","first-page":"11384","article-title":"Shifted chunk transformer for spatio-temporal representational learning","author":"zha","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"ref94","first-page":"2491","article-title":"Associating objects with transformers for video object segmentation","author":"yang","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00718"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00058"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1244"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00803"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00988"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"ref85","first-page":"3719","article-title":"Referring segmentation in images and videos with cross-modal self-attention network","volume":"44","author":"ye","year":"2022","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3082763"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2074"},{"key":"ref135","article-title":"When vision transformers outperform resnets without pretraining or strong data augmentations","author":"chen","year":"2021"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01325"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2018.03.080"},{"key":"ref82","article-title":"Latent video transformer","author":"rakhimov","year":"2020"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00871"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"ref145","first-page":"4117","article-title":"Multi-modal dense video captioning","author":"iashin","year":"2020","journal-title":"Proc IEEE Conf Comput Vis and Pattern Recog"},{"key":"ref84","first-page":"1161","article-title":"Scaling autoregressive video models","author":"weissenborn","year":"2020","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref142","first-page":"68","article-title":"Stand-alone self-attention in vision models","author":"ramachandran","year":"2019","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/ICTAI50040.2020.00176"},{"key":"ref143","first-page":"3965","article-title":"CoAtNet: Marrying convolution and attention for all data sizes","author":"dai","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref140","first-page":"8122","article-title":"Synchronous bidirectional learning for multilingual lip reading","author":"luo","year":"2020","journal-title":"Proc Brit Mach Vis Conf"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01039"},{"key":"ref80","first-page":"214","article-title":"Parameter efficient multimodal transformers for video representation learning","author":"lee","year":"2021","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref78","first-page":"39","article-title":"Summarizing videos with attention","author":"fajtl","year":"2018","journal-title":"Proc Asian Conf Comput Vis"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01151"},{"key":"ref107","first-page":"13352","article-title":"Video instance segmentation using inter-frame communication transformers","author":"hwang","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00911"},{"key":"ref104","first-page":"12493","article-title":"Keeping your eye on the ball: Trajectory attention in video transformers","author":"patrick","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref74","first-page":"8116","article-title":"Support-set bottlenecks for video-text representation learning","author":"patrick","year":"2021","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref105","first-page":"12786","article-title":"TokenLearner: Adaptive space-time tokenization for videos","author":"ryoo","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00289"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.03.026"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2011.157"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01341"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01362"},{"key":"ref70","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","author":"ren","year":"2015","journal-title":"Proc 28th Int Conf Neural Inf Process Syst"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01432"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3113114"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00315"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01322"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00192"},{"key":"ref119","first-page":"698","article-title":"A better use of audio-visual cues: Dense video captioning with bi-modal transformer","author":"iashin","year":"2020","journal-title":"Proc Brit Mach Vis Conf"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00092"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01827"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01564"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_41"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00156"},{"key":"ref115","first-page":"1553","article-title":"Hopper: Multi-hop transformer for spatiotemporal reasoning","author":"zhou","year":"2021","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref63","first-page":"24206","article-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text","author":"akbari","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref116","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00864"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00863"},{"key":"ref113","first-page":"14733","article-title":"UniFormer: Unified transformer for efficient spatial-temporal representation learning","author":"li","year":"2022","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00162"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00194"},{"key":"ref60","first-page":"4651","article-title":"Perceiver: General perception with iterative attention","author":"jaegle","year":"2021","journal-title":"Proc 38th Int Conf Mach Learn"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00063"},{"key":"ref123","first-page":"1403","article-title":"TransformerFusion: Monocular RGB scene reconstruction using transformers","author":"bozic","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_31"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01367"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01332"},{"key":"ref121","first-page":"1086","article-title":"Long short-term transformer for online action detection","author":"xu","year":"2021","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01170"},{"key":"ref169","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01334"},{"key":"ref170","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3416291"},{"key":"ref177","article-title":"Co-training transformer with videos and images improves action recognition","author":"zhang","year":"2021"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.278"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.1016\/S0079-7421(08)60536-8"},{"key":"ref176","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01563"},{"key":"ref173","first-page":"8821","article-title":"Zero-shot text-to-image generation","author":"ramesh","year":"2021","journal-title":"Proc 38th Int Conf Mach Learn"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00186"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00950"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.1038\/4580"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3045007"},{"key":"ref181","article-title":"Watching the world go by: Representation learning from unlabeled videos","author":"gordon","year":"2020"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01232"},{"key":"ref189","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00565"},{"key":"ref186","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2992393"},{"key":"ref187","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.622"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"ref185","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01359"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.320"},{"key":"ref183","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00769"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2019.2923918"},{"key":"ref149","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-66823-5_18"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01004"},{"key":"ref147","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3356051"},{"key":"ref155","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref156","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref153","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00948"},{"key":"ref154","article-title":"Bootstrap your own latent-a new approach to self-supervised learning","author":"grill","year":"2020","journal-title":"Proc 34th Int Conf Neural Inf Process Syst"},{"key":"ref151","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01471"},{"key":"ref152","article-title":"Using self-supervised learning can improve model robustness and uncertainty","author":"hendrycks","year":"2019","journal-title":"Proc 33rd Int Conf Neural Inf Process Syst"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11506"},{"key":"ref159","article-title":"Representation learning with contrastive predictive coding","author":"oord","year":"2018"},{"key":"ref157","article-title":"Self-supervised learning for videos: A survey","author":"schiappa","year":"2022"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01867"},{"key":"ref166","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","author":"chen","year":"2020","journal-title":"Proc 37th Int Conf Mach Learn"},{"key":"ref167","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00331"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref165","first-page":"10268","article-title":"Understanding self-supervised learning dynamics without contrastive pairs","author":"tian","year":"2021","journal-title":"Proc 38th Int Conf Mach Learn"},{"key":"ref162","article-title":"Demystifying contrastive self-supervised learning: Invariances, augmentations and dataset biases","author":"purushwalkam","year":"2020","journal-title":"Proc 34th Int Conf Neural Inf Process Syst"},{"key":"ref163","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref160","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"ref161","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00331"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2022.10.001"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-021-0247-3"},{"key":"ref14","article-title":"AMMUS: A survey of transformer-based pretrained models in natural language processing","author":"kalyan","year":"2021"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00877"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"ref17","article-title":"Efficient transformers: A survey","volume":"55","author":"tay","year":"2020","journal-title":"ACM Comput Surv"},{"key":"ref16","article-title":"A survey of visual transformers","author":"liu","year":"2021"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3505244"},{"key":"ref18","article-title":"A practical survey on faster and lighter transformers","author":"fournier","year":"2021"},{"key":"ref2","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"Proc Conf North Amer Chapter Assoc Comput Linguistics"},{"key":"ref1","first-page":"6000","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc 31st Int Conf Neural Inf Process Syst"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01171"},{"key":"ref192","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01007"},{"key":"ref190","first-page":"2793","article-title":"Attention is not all you need: Pure attention loses rank doubly exponentially with depth","author":"dong","year":"2021","journal-title":"Proc 38th Int Conf Mach Learn"},{"key":"ref199","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1282"},{"key":"ref197","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"ref198","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20729"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i2.20103"},{"key":"ref196","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.244"},{"key":"ref193","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00774"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00713"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10269680\/10041724.pdf?arnumber=10041724","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,9]],"date-time":"2024-07-09T18:27:12Z","timestamp":1720549632000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10041724\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,1]]},"references-count":207,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2023.3243465","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,1]]}}}