{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T03:17:09Z","timestamp":1783135029526,"version":"3.54.6"},"reference-count":95,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100004895","name":"European Social Fund Plus","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004895","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100011929","name":"Programa Operacional Tem\u00e1tico Factores de Competitividade","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100011929","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001871","name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","doi-asserted-by":"publisher","award":["2022.09967"],"award-info":[{"award-number":["2022.09967"]}],"id":[{"id":"10.13039\/501100001871","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001871","name":"Funda\u00e7\u00e3o para a Ci\u00eancia e a Tecnologia","doi-asserted-by":"publisher","award":["18464"],"award-info":[{"award-number":["18464"]}],"id":[{"id":"10.13039\/501100001871","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000780","name":"European Commission","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100000780","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.cviu.2026.104771","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T20:25:57Z","timestamp":1777580757000},"page":"104771","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["JEMA: Joint Embedding of Multimodal and multi-view Alignment in human-centric embedding space for manufacturing"],"prefix":"10.1016","volume":"268","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3879-6908","authenticated-orcid":false,"given":"Jo\u00e3o","family":"Sousa","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Roya","family":"Darabi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Armando","family":"Sousa","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Frank","family":"Brueckner","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu\u00eds Paulo","family":"Reis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ana","family":"Reis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104771_b1","doi-asserted-by":"crossref","unstructured":"Adjogble,\u00a0Franck\u00a0Komi, Warschat,\u00a0Joachim, Hemmje,\u00a0Matthias, 2023. Advanced Intelligent Manufacturing in Process Industry Using Industrial Artificial Intelligence. In: 2023 Portland International Conference on Management of Engineering and Technology. PICMET, pp. 1\u201316.","DOI":"10.23919\/PICMET59654.2023.10216797"},{"key":"10.1016\/j.cviu.2026.104771_b2","first-page":"83","article-title":"Skilled labour shortage in the building construction industry within the central region","volume":"8","author":"Akomah","year":"2020","journal-title":"Balt. J. Real Estate Econ. Constr. Manag."},{"key":"10.1016\/j.cviu.2026.104771_b3","series-title":"Flamingo: a visual language model for few-shot learning","author":"Alayrac","year":"2022"},{"key":"10.1016\/j.cviu.2026.104771_b4","doi-asserted-by":"crossref","unstructured":"Arora,\u00a0Raman, Livescu,\u00a0Karen, 2013. Multi-view CCA-based acoustic features for phonetic recognition across speakers and domains. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing. pp. 7135\u20137139.","DOI":"10.1109\/ICASSP.2013.6639047"},{"key":"10.1016\/j.cviu.2026.104771_b5","series-title":"Self-Supervised learning from images with a Joint-Embedding predictive architecture","author":"Assran","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b6","series-title":"A cookbook of Self-Supervised learning","author":"Balestriero","year":"2023"},{"issue":"2","key":"10.1016\/j.cviu.2026.104771_b7","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","article-title":"Multimodal machine learning: A survey and taxonomy","volume":"41","author":"Baltru\u0161aitis","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104771_b8","doi-asserted-by":"crossref","first-page":"14804","DOI":"10.1109\/ACCESS.2023.3243854","article-title":"A systematic literature review on multimodal machine learning: Applications, challenges, gaps and future directions","volume":"11","author":"Barua","year":"2023","journal-title":"IEEE Access"},{"key":"10.1016\/j.cviu.2026.104771_b9","doi-asserted-by":"crossref","unstructured":"Berroukham,\u00a0Abdelhafid, Housni,\u00a0Khalid, Lahraichi,\u00a0Mohammed, 2023. Vision Transformers: A Review of Architecture, Applications, and Future Directions. In: 2023 7th IEEE Congress on Information Science and Technology (CiSt). pp. 205\u2013210.","DOI":"10.1109\/CiSt56084.2023.10410015"},{"key":"10.1016\/j.cviu.2026.104771_b10","article-title":"The OpenCV library","author":"Bradski","year":"2000","journal-title":"Dr. Dobb\u2019s J. Softw. Tools"},{"key":"10.1016\/j.cviu.2026.104771_b11","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2023.102581","article-title":"Multisensor fusion-based digital twin for localized quality prediction in robotic laser-directed energy deposition","volume":"84","author":"Chen","year":"2023","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.cviu.2026.104771_b12","doi-asserted-by":"crossref","unstructured":"Chopra,\u00a0S., Hadsell,\u00a0R., LeCun,\u00a0Y., 2005. Learning a similarity metric discriminatively, with application to face verification. In: 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition. CVPR\u201905, Vol. 1, pp. 539\u2013546, vol. 1.","DOI":"10.1109\/CVPR.2005.202"},{"key":"10.1016\/j.cviu.2026.104771_b13","series-title":"Introduction to latent variable energy-based models: A path towards autonomous machine intelligence","author":"Dawid","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2020.114060","article-title":"Machine learning and data mining in manufacturing","volume":"166","author":"Dogan","year":"2021","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.cviu.2026.104771_b15","doi-asserted-by":"crossref","unstructured":"Dong,\u00a0Hao, Wang,\u00a0Guodong, Zhang,\u00a0Xinyue, 2022. Aggregation Transformer for Human Pose Estimation. In: 2022 26th International Conference on Pattern Recognition. ICPR, pp. 3660\u20133667.","DOI":"10.1109\/ICPR56361.2022.9956315"},{"key":"10.1016\/j.cviu.2026.104771_b16","series-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"10.1016\/j.cviu.2026.104771_b17","series-title":"Multi-modal alignment using representation codebook","author":"Duan","year":"2022"},{"key":"10.1016\/j.cviu.2026.104771_b18","article-title":"Gen-JEMA: enhanced explainability using generative joint embedding multimodal alignment for monitoring directed energy deposition","author":"Ferreira","year":"2025","journal-title":"J. Intell. Manuf."},{"issue":"4","key":"10.1016\/j.cviu.2026.104771_b19","doi-asserted-by":"crossref","DOI":"10.3390\/met11040672","article-title":"Optimization of direct laser deposition of a martensitic steel powder (Metco 42C) on 42CrMo4 Steel","volume":"11","author":"Ferreira","year":"2021","journal-title":"Metals"},{"key":"10.1016\/j.cviu.2026.104771_b20","series-title":"Advances in Neural Information Processing Systems","article-title":"DeViSE: A deep visual-semantic embedding model","author":"Frome","year":"2013"},{"key":"10.1016\/j.cviu.2026.104771_b21","series-title":"Analyzing and improving representations with the soft nearest neighbor loss","author":"Frosst","year":"2019"},{"key":"10.1016\/j.cviu.2026.104771_b22","doi-asserted-by":"crossref","unstructured":"Galvez,\u00a0Reagan\u00a0L., Bandala,\u00a0Argel\u00a0A., Dadios,\u00a0Elmer\u00a0P., Vicerra,\u00a0Ryan Rhay\u00a0P., Maningo,\u00a0Jose Martin\u00a0Z., 2018. Object Detection Using Convolutional Neural Networks. In: TENCON 2018 - 2018 IEEE Region 10 Conference. pp. 2023\u20132027.","DOI":"10.1109\/TENCON.2018.8650517"},{"key":"10.1016\/j.cviu.2026.104771_b23","series-title":"RVT: Robotic view transformer for 3D object manipulation","author":"Goyal","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b24","series-title":"Bootstrap your own latent: A new approach to self-supervised learning","author":"Grill","year":"2020"},{"issue":"5","key":"10.1016\/j.cviu.2026.104771_b25","doi-asserted-by":"crossref","first-page":"1959","DOI":"10.1007\/s00170-020-05027-0","article-title":"Modeling of the laser powder\u2013based directed energy deposition process for additive manufacturing: a review","volume":"107","author":"Guan","year":"2020","journal-title":"Int. J. Adv. Manuf. Technol."},{"key":"10.1016\/j.cviu.2026.104771_b26","doi-asserted-by":"crossref","unstructured":"Gulsoy,\u00a0Tolgahan, Kablan,\u00a0Elif\u00a0Baykal, 2023. Diagnosis of lung cancer based on CT scans using Vision Transformers. In: 2023 14th International Conference on Electrical and Electronics Engineering. ELECO, pp. 1\u20135.","DOI":"10.1109\/ELECO60389.2023.10416046"},{"key":"10.1016\/j.cviu.2026.104771_b27","doi-asserted-by":"crossref","first-page":"63373","DOI":"10.1109\/ACCESS.2019.2916887","article-title":"Deep multimodal representation learning: A survey","volume":"7","author":"Guo","year":"2019","journal-title":"IEEE Access"},{"issue":"1","key":"10.1016\/j.cviu.2026.104771_b28","doi-asserted-by":"crossref","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","article-title":"A survey on vision transformer","volume":"45","author":"Han","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104771_b29","series-title":"What makes Multi-modal learning better than single (Provably)","author":"Huang","year":"2021"},{"key":"10.1016\/j.cviu.2026.104771_b30","series-title":"Comprehensive process-molten pool relations modeling using CNN for wire-feed laser additive manufacturing","author":"Jamnikar","year":"2021"},{"issue":"1","key":"10.1016\/j.cviu.2026.104771_b31","doi-asserted-by":"crossref","first-page":"903","DOI":"10.1007\/s00170-022-09248-3","article-title":"In-process comprehensive prediction of bead geometry for laser wire-feed DED system using molten pool sensing data and multi-modality CNN","volume":"121","author":"Jamnikar","year":"2022","journal-title":"Int. J. Adv. Manuf. Technol."},{"key":"10.1016\/j.cviu.2026.104771_b32","doi-asserted-by":"crossref","first-page":"803","DOI":"10.1016\/j.jmapro.2022.05.013","article-title":"In situ microstructure property prediction by modeling molten pool-quality relations for wire-feed laser additive manufacturing","volume":"79","author":"Jamnikar","year":"2022","journal-title":"J. Manuf. Process."},{"key":"10.1016\/j.cviu.2026.104771_b33","series-title":"Scaling Up visual and Vision-Language representation learning with noisy text supervision","author":"Jia","year":"2021"},{"key":"10.1016\/j.cviu.2026.104771_b34","article-title":"M2FNet: Multi-modal fusion network for object detection from visible and thermal infrared images","volume":"130","author":"Jiang","year":"2024","journal-title":"Int. J. Appl. Earth Obs. Geoinf."},{"key":"10.1016\/j.cviu.2026.104771_b35","doi-asserted-by":"crossref","unstructured":"Jmour,\u00a0Nadia, Zayen,\u00a0Sehla, Abdelkrim,\u00a0Afef, 2018. Convolutional neural networks for image classification. In: 2018 International Conference on Advanced Systems and Electric Technologies. pp. 397\u2013402.","DOI":"10.1109\/ASET.2018.8379889"},{"key":"10.1016\/j.cviu.2026.104771_b36","article-title":"Review of the construction labour demand and shortages in the EU","author":"Juricic","year":"2021","journal-title":"Buildings"},{"key":"10.1016\/j.cviu.2026.104771_b37","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123640","article-title":"Video2Music: Suitable music generation from videos using an Affective Multimodal Transformer model","volume":"249","author":"Kang","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.cviu.2026.104771_b38","doi-asserted-by":"crossref","first-page":"11424","DOI":"10.1016\/j.jmrt.2020.08.039","article-title":"Additive manufacturing method and different welding applications","volume":"9","author":"Karayel","year":"2020","journal-title":"J. Mater. Res. Technol."},{"key":"10.1016\/j.cviu.2026.104771_b39","series-title":"Deep visual-semantic alignments for generating image descriptions","author":"Karpathy","year":"2015"},{"key":"10.1016\/j.cviu.2026.104771_b40","series-title":"Time2Vec: Learning a vector representation of time","author":"Kazemi","year":"2019"},{"key":"10.1016\/j.cviu.2026.104771_b41","doi-asserted-by":"crossref","first-page":"193907","DOI":"10.1109\/ACCESS.2020.3031549","article-title":"Contrastive representation learning: A framework and review","volume":"8","author":"Kh\u00e1c","year":"2020","journal-title":"IEEE Access"},{"issue":"10s","key":"10.1016\/j.cviu.2026.104771_b42","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3505244","article-title":"Transformers in vision: A survey","volume":"54","author":"Khan","year":"2022","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.cviu.2026.104771_b43","series-title":"Supervised contrastive learning","author":"Khosla","year":"2021"},{"key":"10.1016\/j.cviu.2026.104771_b44","doi-asserted-by":"crossref","DOI":"10.1016\/j.optlaseng.2023.107661","article-title":"Infrared thermographic imaging based real-time layer height estimation during directed energy deposition","volume":"168","author":"Kim","year":"2023","journal-title":"Opt. Lasers Eng."},{"key":"10.1016\/j.cviu.2026.104771_b45","series-title":"Multimodality of AI for education: Towards artificial general intelligence","author":"Lee","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b46","doi-asserted-by":"crossref","unstructured":"Li,\u00a0Chen, Kaszowska,\u00a0Aleksandra, Chrysostomou,\u00a0Dimitrios, 2023. A Multimodal Attention Tracking in Human-Robot Interaction in Industrial Robots for Manufacturing Tasks. In: 2023 28th International Conference on Automation and Computing. ICAC, pp. 1\u20135.","DOI":"10.1109\/ICAC57885.2023.10275168"},{"key":"10.1016\/j.cviu.2026.104771_b47","series-title":"Foundations and trends in multimodal machine learning: Principles, challenges, and open questions","author":"Liang","year":"2023"},{"issue":"3","key":"10.1016\/j.cviu.2026.104771_b48","doi-asserted-by":"crossref","first-page":"1319","DOI":"10.1109\/TPAMI.2023.3341723","article-title":"Editorial: Learning with fewer labels in computer vision","volume":"46","author":"Liu","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104771_b49","series-title":"ViT-A*: Legged robot path planning using vision transformer A*","author":"Liu","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b50","doi-asserted-by":"crossref","unstructured":"Lv,\u00a0Pengyuan, Wu,\u00a0Wenjun, Zhong,\u00a0Yanfei, Zhang,\u00a0Liangpei, 2022. Review of Vision Transformer Models for Remote Sensing Image Scene Classification. In: IGARSS 2022 - 2022 IEEE International Geoscience and Remote Sensing Symposium. pp. 2231\u20132234.","DOI":"10.1109\/IGARSS46834.2022.9883054"},{"issue":"66","key":"10.1016\/j.cviu.2026.104771_b51","doi-asserted-by":"crossref","first-page":"eabm6074","DOI":"10.1126\/scirobotics.abm6074","article-title":"Robot operating system 2: Design, architecture, and uses in the wild","volume":"7","author":"Macenski","year":"2022","journal-title":"Sci. Robot."},{"key":"10.1016\/j.cviu.2026.104771_b52","series-title":"Processing parameters in laser metal deposition process","first-page":"61","author":"Mahamood","year":"2018"},{"key":"10.1016\/j.cviu.2026.104771_b53","doi-asserted-by":"crossref","unstructured":"Mahmud,\u00a0Bahar\u00a0Uddin, Hong,\u00a0Guan\u00a0Yue, 2022. Semantic Image Segmentation using CNN (Convolutional Neural Network) based Technique. In: 2022 IEEE World Conference on Applied Intelligence and Computing. AIC, pp. 210\u2013214.","DOI":"10.1109\/AIC55036.2022.9848977"},{"key":"10.1016\/j.cviu.2026.104771_b54","series-title":"The more you know: Using knowledge graphs for image classification","author":"Marino","year":"2017"},{"issue":"2","key":"10.1016\/j.cviu.2026.104771_b55","doi-asserted-by":"crossref","first-page":"494","DOI":"10.3390\/s22020494","article-title":"A Physics-Informed convolutional neural network with custom loss functions for porosity prediction in laser metal deposition","volume":"22","author":"McGowan","year":"2022","journal-title":"Sensors"},{"issue":"6","key":"10.1016\/j.cviu.2026.104771_b56","doi-asserted-by":"crossref","first-page":"1179","DOI":"10.1109\/JSTSP.2022.3207050","article-title":"Self-Supervised speech representation learning: A review","volume":"16","author":"Mohamed","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.cviu.2026.104771_b57","doi-asserted-by":"crossref","unstructured":"Mohan,\u00a0Paul, Brilley,\u00a0Batley\u00a0C., Kumar A.\u00a0A.,\u00a0Nippun, 2022. Multimodal Representation Learning:Cross-modality and Shared Representation. In: 2022 International Conference on Industry 4.0 Technology (I4Tech). pp. 1\u20135.","DOI":"10.1109\/I4Tech55392.2022.9952528"},{"key":"10.1016\/j.cviu.2026.104771_b58","series-title":"Multimodal transfer deep learning with applications in Audio-Visual recognition","author":"Moon","year":"2016"},{"issue":"4","key":"10.1016\/j.cviu.2026.104771_b59","doi-asserted-by":"crossref","first-page":"3005","DOI":"10.1007\/s10462-022-10246-w","article-title":"Human-in-the-loop machine learning: a state of the art","volume":"56","author":"Mosqueira-Rey","year":"2022","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.cviu.2026.104771_b60","doi-asserted-by":"crossref","DOI":"10.1016\/j.jmatprotec.2021.117485","article-title":"Mechanistic artificial intelligence (mechanistic-AI) for modeling, design, and control of advanced manufacturing processes: Current state and perspectives","volume":"302","author":"Mozaffar","year":"2022","journal-title":"J. Mater. Process. Technol."},{"key":"10.1016\/j.cviu.2026.104771_b61","series-title":"Representation learning with contrastive predictive coding","author":"van\u00a0den Oord","year":"2019"},{"key":"10.1016\/j.cviu.2026.104771_b62","doi-asserted-by":"crossref","first-page":"1064","DOI":"10.1016\/j.jmapro.2022.07.033","article-title":"In situ quality monitoring in direct energy deposition process using co-axial process zone imaging and deep contrastive learning","volume":"81","author":"Pandiyan","year":"2022","journal-title":"J. Manuf. Process."},{"key":"10.1016\/j.cviu.2026.104771_b63","article-title":"Real-time monitoring and quality assurance for laser-based directed energy deposition: integrating co-axial imaging and self-supervised deep learning framework","author":"Pandiyan","year":"2023","journal-title":"J. Intell. Manuf."},{"key":"10.1016\/j.cviu.2026.104771_b64","first-page":"2825","article-title":"Scikit-learn: Machine learning in Python","volume":"12","author":"Pedregosa","year":"2011","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.cviu.2026.104771_b65","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2022.102445","article-title":"Track geometry prediction for Laser Metal Deposition based on on-line artificial vision and deep neural networks","volume":"79","author":"Perani","year":"2023","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.cviu.2026.104771_b66","doi-asserted-by":"crossref","DOI":"10.3389\/frai.2023.1156630","article-title":"Long-short term memory networks for modeling track geometry in laser metal deposition","volume":"6","author":"Perani","year":"2023","journal-title":"Front. Artif. Intell."},{"key":"10.1016\/j.cviu.2026.104771_b67","series-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.cviu.2026.104771_b68","unstructured":"Reed,\u00a0Scott, Zolna,\u00a0Konrad, Parisotto,\u00a0Emilio, Colmenarejo,\u00a0Sergio\u00a0Gomez, Novikov,\u00a0Alexander, Barth-Maron,\u00a0Gabriel, Gimenez,\u00a0Mai, Sulsky,\u00a0Yury, Kay,\u00a0Jackie, Springenberg,\u00a0Jost\u00a0Tobias, Eccles,\u00a0Tom, Bruce,\u00a0Jake, Razavi,\u00a0Ali, Edwards,\u00a0Ashley, Heess,\u00a0Nicolas, Chen,\u00a0Yutian, Hadsell,\u00a0Raia, Vinyals,\u00a0Oriol, Bordbar,\u00a0Mahyar, de\u00a0Freitas,\u00a0Nando, 2022. A Generalist Agent."},{"issue":"1","key":"10.1016\/j.cviu.2026.104771_b69","doi-asserted-by":"crossref","DOI":"10.1038\/s41598-024-52821-x","article-title":"Wildfire spreading prediction using multimodal data and deep neural network approach","volume":"14","author":"Shadrin","year":"2024","journal-title":"Sci. Rep."},{"issue":"1","key":"10.1016\/j.cviu.2026.104771_b70","doi-asserted-by":"crossref","first-page":"173","DOI":"10.1109\/JBHI.2017.2655720","article-title":"Multimodal neuroimaging feature learning with multimodal stacked deep polynomial networks for diagnosis of Alzheimer\u2019s Disease","volume":"22","author":"Shi","year":"2018","journal-title":"IEEE J. Biomed. Health Inform."},{"issue":"3","key":"10.1016\/j.cviu.2026.104771_b71","article-title":"To compress or not to compress\u2014Self-supervised learning and information theory: A review","volume":"26","author":"Shwartz\u00a0Ziv","year":"2024","journal-title":"Entropy"},{"issue":"2","key":"10.1016\/j.cviu.2026.104771_b72","doi-asserted-by":"crossref","first-page":"97","DOI":"10.1080\/17686733.2020.1829889","article-title":"In-situ monitoring of a laser metal deposition (LMD) process: comparison of MWIR, SWIR and high-speed NIR thermography","volume":"19","author":"Simon J.\u00a0Altenburg","year":"2022","journal-title":"Quant. InfraRed Thermogr. J."},{"key":"10.1016\/j.cviu.2026.104771_b73","doi-asserted-by":"crossref","unstructured":"Sisca,\u00a0Francesco\u00a0G., Angioletti,\u00a0Cecilia\u00a0M., Taisch,\u00a0Marco, Colwill,\u00a0James\u00a0A., 2016. Additive manufacturing as a strategic tool for industrial competition. In: 2016 IEEE 2nd International Forum on Research and Technologies for Society and Industry Leveraging a Better Tomorrow. RTSI, pp. 1\u20137.","DOI":"10.1109\/RTSI.2016.7740609"},{"key":"10.1016\/j.cviu.2026.104771_b74","series-title":"A review of explainable artificial intelligence in manufacturing","author":"Sofianidis","year":"2021"},{"key":"10.1016\/j.cviu.2026.104771_b75","doi-asserted-by":"crossref","first-page":"30845","DOI":"10.1109\/ACCESS.2025.3537859","article-title":"Artificial intelligence for control in Laser-Based additive manufacturing: A systematic review","volume":"13","author":"Sousa","year":"2025","journal-title":"IEEE Access"},{"key":"10.1016\/j.cviu.2026.104771_b76","article-title":"JEMA-SINDYc: End-to-end control using joint embedding multimodal alignment in directed energy deposition","volume":"109","author":"Sousa","year":"2025","journal-title":"Addit. Manuf."},{"key":"10.1016\/j.cviu.2026.104771_b77","doi-asserted-by":"crossref","DOI":"10.1016\/j.rcim.2024.102892","article-title":"Human-in-the-loop Multi-objective Bayesian optimization for directed energy deposition with in-situ monitoring","volume":"92","author":"Sousa","year":"2025","journal-title":"Robot. Comput.-Integr. Manuf."},{"key":"10.1016\/j.cviu.2026.104771_b78","doi-asserted-by":"crossref","unstructured":"Sun,\u00a0YuTong, Cheng,\u00a0Dalei, Chen,\u00a0Yaxi, He,\u00a0Zhiwei, 2023. DynamicMBFN: Dynamic Multimodal Bottleneck Fusion Network for Multimodal Emotion Recognition. In: 2023 3rd International Symposium on Computer Technology and Information Science. ISCTIS, pp. 639\u2013644.","DOI":"10.1109\/ISCTIS58954.2023.10213035"},{"key":"10.1016\/j.cviu.2026.104771_b79","doi-asserted-by":"crossref","first-page":"271","DOI":"10.1016\/j.mattod.2021.03.020","article-title":"Directed energy deposition (DED) additive manufacturing: Physical characteristics, defects, challenges and applications","volume":"49","author":"Svetlizky","year":"2021","journal-title":"Mater. Today"},{"issue":"11","key":"10.1016\/j.cviu.2026.104771_b80","doi-asserted-by":"crossref","first-page":"3437","DOI":"10.1007\/s00170-020-05569-3","article-title":"A review on in situ monitoring technology for directed energy deposition of metals","volume":"108","author":"Tang","year":"2020","journal-title":"Int. J. Adv. Manuf. Technol."},{"key":"10.1016\/j.cviu.2026.104771_b81","doi-asserted-by":"crossref","unstructured":"Teh,\u00a0Shem, Sivakumar,\u00a0Saaveethya, Motalebi,\u00a0Foad, 2024. Vision Transformers for Biomedical Applications *. In: 2024 International Conference on Green Energy, Computing and Sustainable Technology. GECOST, pp. 195\u2013201.","DOI":"10.1109\/GECOST60902.2024.10474871"},{"key":"10.1016\/j.cviu.2026.104771_b82","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123153","article-title":"Determining the onset of driver\u2019s preparatory action for take-over in automated driving using multimodal data","volume":"246","author":"Teshima","year":"2024","journal-title":"Expert Syst. Appl."},{"issue":"41011","key":"10.1016\/j.cviu.2026.104771_b83","article-title":"Deep learning-based data fusion method for in situ porosity detection in laser-based additive manufacturing","volume":"143","author":"Tian","year":"2020","journal-title":"J. Manuf. Sci. Eng."},{"issue":"1","key":"10.1016\/j.cviu.2026.104771_b84","doi-asserted-by":"crossref","first-page":"72","DOI":"10.1038\/s41698-024-00573-2","article-title":"Large language models and multimodal foundation models for precision oncology","volume":"8","author":"Truhn","year":"2024","journal-title":"Npj Precis. Oncol."},{"key":"10.1016\/j.cviu.2026.104771_b85","series-title":"Attention is all you need","author":"Vaswani","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b86","series-title":"Image as a Foreign Language: BEiT pretraining for all vision and Vision-Language tasks","author":"Wang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104771_b87","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102370","article-title":"Joint Semantic Segmentation using representations of LiDAR point clouds and camera images","volume":"108","author":"Wu","year":"2024","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.cviu.2026.104771_b88","series-title":"Dynamic memory networks for visual and textual question answering","author":"Xiong","year":"2016"},{"issue":"10","key":"10.1016\/j.cviu.2026.104771_b89","doi-asserted-by":"crossref","first-page":"12113","DOI":"10.1109\/TPAMI.2023.3275156","article-title":"Multimodal learning with transformers: A survey","volume":"45","author":"Xu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104771_b90","first-page":"1","article-title":"Predictions of in-situ melt pool geometric signatures via machine learning techniques for laser metal deposition","volume":"00","author":"Ye","year":"2022","journal-title":"Int. J. Comput. Integr. Manuf."},{"key":"10.1016\/j.cviu.2026.104771_b91","series-title":"Video paragraph captioning using hierarchical recurrent neural networks","author":"Yu","year":"2016"},{"key":"10.1016\/j.cviu.2026.104771_b92","doi-asserted-by":"crossref","first-page":"188","DOI":"10.1016\/j.inffus.2020.06.001","article-title":"Foundations of multimodal Co-learning","volume":"64","author":"Zadeh","year":"2020","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.cviu.2026.104771_b93","series-title":"Advances in Neural Information Processing Systems","first-page":"17882","article-title":"Rank-N-Contrast: Learning continuous representations for regression","volume":"Vol. 36","author":"Zha","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b94","series-title":"Meta-Transformer: A unified framework for multimodal learning","author":"Zhang","year":"2023"},{"key":"10.1016\/j.cviu.2026.104771_b95","first-page":"1","article-title":"Deep multimodal data fusion","volume":"56","author":"Zhao","year":"2024","journal-title":"ACM Comput. Surv."}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001384?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001384?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T02:28:48Z","timestamp":1783132128000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001384"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":95,"alternative-id":["S1077314226001384"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104771","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"JEMA: Joint Embedding of Multimodal and multi-view Alignment in human-centric embedding space for manufacturing","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104771","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104771"}}