{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T08:05:04Z","timestamp":1773734704591,"version":"3.50.1"},"reference-count":42,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T00:00:00Z","timestamp":1763424000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T00:00:00Z","timestamp":1763424000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,11,18]]},"DOI":"10.1109\/itsc60802.2025.11423488","type":"proceedings-article","created":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T20:10:23Z","timestamp":1773691823000},"page":"3843-3850","source":"Crossref","is-referenced-by-count":0,"title":["SMAE-DIM: Vehicle-Centric Semantic Masked AutoEncoders Pre-Training by Distilling Multimodal Foundational Models"],"prefix":"10.1109","author":[{"given":"Alexandre","family":"Marques","sequence":"first","affiliation":[{"name":"University of Coimbra,Institute of Systems and Robotics,Coimbra,Portugal"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pedro","family":"Ferreira","sequence":"additional","affiliation":[{"name":"University of Coimbra,Institute of Systems and Robotics,Coimbra,Portugal"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bruno","family":"Silva","sequence":"additional","affiliation":[{"name":"University of Coimbra,Institute of Systems and Robotics,Coimbra,Portugal"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jorge","family":"Batista","sequence":"additional","affiliation":[{"name":"University of Coimbra,Institute of Systems and Robotics,Dept. of Electrical and Computer Engineering,Coimbra,Portugal"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref2","author":"Bao","year":"2022","journal-title":"Beit: Bert pre-training of image transformers"},{"key":"ref3","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021","journal-title":"in ICML. PMLR"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3178144"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01474"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00558"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19967"},{"key":"ref9","author":"Sun","year":"2023","journal-title":"Eva-clip: Improved training techniques for clip at scale"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01423"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73235-5_25"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25130"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01898-4"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/wacv61041.2025.00091"},{"key":"ref16","author":"Chen","year":"2020","journal-title":"A simple framework for contrastive learning of visual representations"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.00950"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72658-3_9"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3336525"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i4.28078"},{"key":"ref21","author":"Jiang","year":"2023","journal-title":"Layer grafted pre-training: Bridging contrastive learning and masked image modeling for label-efficient representations"},{"key":"ref22","author":"Li","year":"2022","journal-title":"Supervision exists everywhere: A data efficient contrastive languageimage pre-training paradigm"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19809-0_30"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01857"},{"key":"ref25","author":"Peng","journal-title":"BEiT v2: Masked Image Modeling with Vector-Quantized Visual Tokenizers"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref27","author":"Zhou","year":"2021","journal-title":"ibot: Image bert pre-training with online tokenizer"},{"key":"ref28","author":"Alkin","year":"2025","journal-title":"Mim-refiner: A contrastive learning boost from intermediate pre-trained representations"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i6.28373"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref31","author":"Yu","year":"2015","journal-title":"Lsun: Construction of a large-scale image dataset using deep learning with humans in the loop"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.3390\/app10144913"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2796240"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01413"},{"key":"ref35","author":"Vo","year":"2024","journal-title":"Automatic data curation for self-supervised learning: A clustering-based approach"},{"key":"ref36","author":"Oquab","year":"2023","journal-title":"Dinov2: Learning robust visual features without supervision"},{"key":"ref37","author":"Liu","year":"2024","journal-title":"Llava-next: Improved reasoning, ocr, and world knowledge"},{"key":"ref38","author":"Devlin","year":"2019","journal-title":"Bert: Pre-training of deep bidirectional transformers for language understanding"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2024.105171"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_53"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.77"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413578"}],"event":{"name":"2025 IEEE 28th International Conference on Intelligent Transportation Systems (ITSC)","location":"Gold Coast, Australia","start":{"date-parts":[[2025,11,18]]},"end":{"date-parts":[[2025,11,21]]}},"container-title":["2025 IEEE 28th International Conference on Intelligent Transportation Systems (ITSC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11422813\/11423000\/11423488.pdf?arnumber=11423488","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T05:48:58Z","timestamp":1773726538000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11423488\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,18]]},"references-count":42,"URL":"https:\/\/doi.org\/10.1109\/itsc60802.2025.11423488","relation":{},"subject":[],"published":{"date-parts":[[2025,11,18]]}}}