{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,16]],"date-time":"2025-09-16T16:24:56Z","timestamp":1758039896336,"version":"3.44.0"},"reference-count":37,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/icmew68306.2025.11152190","type":"proceedings-article","created":{"date-parts":[[2025,9,10]],"date-time":"2025-09-10T17:41:25Z","timestamp":1757526085000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["MVLLaVA: An Intelligent Agent for Unified and Flexible Novel View Synthesis"],"prefix":"10.1109","author":[{"given":"Hanyu","family":"Jiang","sequence":"first","affiliation":[{"name":"University of Chinese Academy of Sciences,School of Engineering Science,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Xue","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences,School of Engineering Science,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Lan","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences,School of Engineering Science,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guohong","family":"Hu","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences,School of Engineering Science,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ke","family":"Lu","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory,Shenzhen,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01018"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00749"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"article-title":"Advances in 3d generation: A survey","year":"2024","author":"Li","key":"ref5"},{"article-title":"Novel view synthesis with diffusion models","year":"2022","author":"Watson","key":"ref6"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00853"},{"article-title":"Zero123++: a single image to consistent multi-view diffusion base model","year":"2023","author":"Shi","key":"ref8"},{"article-title":"Mvdream: Multi-view diffusion for 3d generation","year":"2023","author":"Shi","key":"ref9"},{"article-title":"Imagedream: Image-prompt multi-view diffusion for 3d generation","year":"2023","author":"Wang","key":"ref10"},{"article-title":"Cat3d: Create anything in 3d with multi-view diffusion models","year":"2024","author":"Gao","key":"ref11"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref13","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Advances in neural information processing systems"},{"key":"ref14","first-page":"2256","article-title":"Deep unsupervised learning using nonequilibrium thermodynamics","volume-title":"International conference on machine learning.","author":"Sohl-Dickstein","year":"2015"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00286"},{"article-title":"Direct3d: Scalable image-to-3d generation via 3d latent diffusion transformer","year":"2024","author":"Wu","key":"ref16"},{"key":"ref17","article-title":"Scene representation networks: Continuous 3d-structure-aware neural scene representations","volume":"32","author":"Sitzmann","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Instruction tuning with gpt-4","year":"2023","author":"Peng","key":"ref18"},{"volume-title":"Chatgpt blog","year":"2023","key":"ref19"},{"article-title":"Gpt-4 technical report","year":"2023","author":"Achiam","key":"ref20"},{"article-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","year":"2023","author":"Zhu","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/tmm.2025.3557704"},{"key":"ref23","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"International conference on machine learning.","author":"Li","year":"2023"},{"article-title":"Emu: Generative pretraining in multimodality","year":"2023","author":"Sun","key":"ref24"},{"key":"ref25","first-page":"21487","article-title":"Generating images with multimodal language models","volume":"36","author":"Koh","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Mastering text-to-image diffusion: Recaptioning, planning, and generating with multimodal llms","volume-title":"Forty-first International Conference on Machine Learning","author":"Yang","key":"ref26"},{"article-title":"Lora: Low-rank adaptation of large language models","year":"2021","author":"Hu","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"ref29","article-title":"Scalable 3d captioning with pretrained models","volume":"36","author":"Luo","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"journal-title":"Judging llm-as-a-judge with mt-bench and chatbot arena","year":"2023","author":"Zheng","key":"ref30"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"ref32","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning.","author":"Radford","year":"2021"},{"year":"2024","key":"ref33","article-title":"Hello gpt-4o"},{"year":"2024","key":"ref34","article-title":"Introducing claude 3.5 sonnet"},{"issue":"8","key":"ref35","article-title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond. arxiv 2023","volume":"1","author":"Bai","year":"2023"},{"article-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","year":"2024","author":"Wang","key":"ref36"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811809"}],"event":{"name":"2025 IEEE International Conference on Multimedia and Expo Workshops (ICMEW)","start":{"date-parts":[[2025,6,30]]},"location":"Nantes, France","end":{"date-parts":[[2025,7,4]]}},"container-title":["2025 IEEE International Conference on Multimedia and Expo Workshops (ICMEW)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11152022\/11152034\/11152190.pdf?arnumber=11152190","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T04:47:51Z","timestamp":1757566071000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11152190\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":37,"URL":"https:\/\/doi.org\/10.1109\/icmew68306.2025.11152190","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}