{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T12:05:25Z","timestamp":1777637125788,"version":"3.51.4"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T00:00:00Z","timestamp":1768348800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T00:00:00Z","timestamp":1768348800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["RS-2025-02217071"],"award-info":[{"award-number":["RS-2025-02217071"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,1,14]]},"DOI":"10.1109\/icoin68469.2026.11480517","type":"proceedings-article","created":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T19:45:18Z","timestamp":1777405518000},"page":"971-974","source":"Crossref","is-referenced-by-count":0,"title":["Text-Centric Multimodal Alignment via Dual-Level Optimization"],"prefix":"10.1109","author":[{"given":"Jin","family":"Hong","sequence":"first","affiliation":[{"name":"Graduate School of Artificial Intelligence, Chung-Ang University,Seoul,South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"JuHyeon","family":"Park","sequence":"additional","affiliation":[{"name":"Graduate School of Artificial Intelligence, Chung-Ang University,Seoul,South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"JunSeok","family":"Kwon","sequence":"additional","affiliation":[{"name":"Chung-Ang University,Department of Software Engineering,Seoul,South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Learning transferable visual models from natural language supervision","volume-title":"ICML","author":"Radford","year":"2021"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095969"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747669"},{"key":"ref5","article-title":"VATT: Transformers for multimodal self-supervised learning from raw video, audio and text","volume-title":"NeurIPS","author":"Akbari","year":"2021"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"ref7","article-title":"LanguageBind: Extending video-language pretraining to N-modality by language-based semantic alignment","volume-title":"ICLR","author":"Zhu","year":"2024"},{"key":"ref8","volume-title":"Representation learning with contrastive predictive coding","author":"Van Den Oord","year":"2018"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/7503.003.0069"},{"key":"ref10","first-page":"723773","article-title":"A kernel two-sample test","volume":"13","author":"Gretton","year":"2012","journal-title":"JMLR"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"ref12","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume-title":"ICML","author":"Jia","year":"2021"},{"key":"ref13","volume-title":"Florence: A new foundation model for computer vision","author":"Yuan","year":"2021"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19809-0_30"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"ref16","article-title":"LAION-5B: An open large-scale dataset for training next generation image-text models","volume-title":"NeurIPS","author":"Schuhmann","year":"2022"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00120"},{"key":"ref18","volume-title":"Point-Bind & Point-LLM: Aligning point cloud with multi-modality for 3D understanding, generation, and instruction following","author":"Guo","year":"2023"},{"key":"ref19","article-title":"Learning transferable features with deep adaptation networks","volume-title":"ICML","author":"Long","year":"2015"},{"key":"ref20","volume-title":"Deep domain confusion: Maximizing for domain invariance","author":"Tzeng","year":"2014"},{"key":"ref21","article-title":"MMD GAN: Towards deeper understanding of moment matching network","volume-title":"NeurIPS","author":"Li","year":"2017"},{"key":"ref22","volume-title":"Representation learning with contrastive predictive coding","author":"Van Den Oord","year":"2018"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1126\/science.aab3050"},{"key":"ref24","article-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","volume-title":"ICLR","author":"Dosovitskiy","year":"2021"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746312"},{"key":"ref26","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"ICML","author":"Houlsby","year":"2019"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.7"},{"key":"ref28","volume-title":"Layer normalization","author":"Ba","year":"2016"},{"key":"ref29","article-title":"Temporal-aware acoustic-to-semantic mappings for audio-visual spatialization","volume-title":"WACV","author":"Yuan","year":"2023"}],"event":{"name":"2026 40th International Conference on Information Networking (ICOIN)","location":"Hanoi, Vietnam","start":{"date-parts":[[2026,1,14]]},"end":{"date-parts":[[2026,1,16]]}},"container-title":["2026 40th International Conference on Information Networking (ICOIN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11480424\/11480459\/11480517.pdf?arnumber=11480517","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T06:02:39Z","timestamp":1777442559000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11480517\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,14]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icoin68469.2026.11480517","relation":{},"subject":[],"published":{"date-parts":[[2026,1,14]]}}}