{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T20:03:26Z","timestamp":1776888206087,"version":"3.51.2"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005047","name":"Natural Science Foundation of Liaoning Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005047","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,4,6]]},"DOI":"10.1109\/icassp49660.2025.10889541","type":"proceedings-article","created":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T13:52:43Z","timestamp":1741787563000},"page":"1-5","source":"Crossref","is-referenced-by-count":3,"title":["Boosting Text-To-Image Generation via Multilingual Prompting in Large Multimodal Models"],"prefix":"10.1109","author":[{"given":"Yongyu","family":"Mu","sequence":"first","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hengyu","family":"Li","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junxin","family":"Wang","sequence":"additional","affiliation":[{"name":"Dalian Jiaotong University,School of Economics and Management,Dalian,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoxuan","family":"Zhou","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenglong","family":"Wang","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yingfeng","family":"Luo","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiaozhi","family":"He","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tong","family":"Xiao","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guocheng","family":"Chen","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingbo","family":"Zhu","sequence":"additional","affiliation":[{"name":"Northeastern University,School of Computer Science and Engineering,Shenyang,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"16 784","article-title":"GLIDE: towards photorealistic image generation and editing with text-guided diffusion models","volume-title":"International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA","volume":"162","author":"Nichol"},{"key":"ref2","first-page":"8821","article-title":"Zero-shot text-to-image generation","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event","volume":"139","author":"Ramcsh"},{"key":"ref3","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022","author":"Saharia"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.807"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3501825"},{"key":"ref6","article-title":"Generative multimodal models are in-context learners","author":"Sun","year":"2023","journal-title":"CoRR"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00671"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.826"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.163"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.379"},{"key":"ref11","article-title":"Large language models are parallel multilingual learners","author":"Mu","year":"2024","journal-title":"CoRR"},{"key":"ref12","article-title":"LAION-5B: an open large-scale dataset for training next generation image-text models","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022","author":"Schuhmann"},{"key":"ref13","article-title":"Denoising diffusion probabilistic models","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual","author":"Ho"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"issue":"3","key":"ref17","first-page":"8","article-title":"Improving image generation with better captions","volume-title":"Computer Science","volume":"2","author":"Betker","year":"2023"},{"key":"ref18","article-title":"T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","author":"Huang"},{"key":"ref19","article-title":"Magicbrush: A manually annotated dataset for instruction-guided image editing","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","author":"Zhang"},{"key":"ref20","first-page":"12888","article-title":"BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA","volume":"162","author":"Li"},{"key":"ref21","article-title":"Imagereward: Learning and evaluating human preferences for text-to-image generation","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","author":"Xu"},{"key":"ref22","article-title":"Lumina-next: Making lumina-t2x stronger and faster with next-dit","author":"Zhuo","year":"2024","journal-title":"CoRR"},{"key":"ref23","article-title":"Gemma: Open models based on gemini research and technology","author":"Mesnard","year":"2024","journal-title":"CoRR"},{"key":"ref24","article-title":"Optimizing prompts for text-to-image generation","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","author":"Hao"},{"key":"ref25","article-title":"Aligner: Achieving efficient alignment through weak-to-strong correction","author":"Ji","year":"2024","journal-title":"CoRR"},{"key":"ref26","article-title":"Weak-to-strong generalization: Eliciting strong capabilities with weak supervision","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024","author":"Bums"}],"event":{"name":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Hyderabad, India","start":{"date-parts":[[2025,4,6]]},"end":{"date-parts":[[2025,4,11]]}},"container-title":["ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10887540\/10887541\/10889541.pdf?arnumber=10889541","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T05:24:42Z","timestamp":1774416282000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10889541\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,6]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/icassp49660.2025.10889541","relation":{},"subject":[],"published":{"date-parts":[[2025,4,6]]}}}