{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:05:08Z","timestamp":1784268308493,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFB36066"],"award-info":[{"award-number":["2022YFB36066"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shenzhen Science and Technology Project","award":["KJZD20240903103210014, JCYJ20220818101001004"],"award-info":[{"award-number":["KJZD20240903103210014, JCYJ20220818101001004"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754964","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"3047-3056","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["SLGaussian: Fast Language Gaussian Splatting in Sparse Views"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-1002-015X","authenticated-orcid":false,"given":"Kangjie","family":"Chen","sequence":"first","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8546-5529","authenticated-orcid":false,"given":"BingQuan","family":"Dai","sequence":"additional","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8304-5353","authenticated-orcid":false,"given":"Minghan","family":"Qin","sequence":"additional","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3919-9827","authenticated-orcid":false,"given":"Dongbin","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2352-8807","authenticated-orcid":false,"given":"Peihao","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2881-0518","authenticated-orcid":false,"given":"Yingshuang","family":"Zou","sequence":"additional","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2792-8469","authenticated-orcid":false,"given":"Haoqian","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua Shenzhen International Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00580"},{"key":"e_1_3_2_1_2_1","volume-title":"Segment Any 3D Gaussians. arXiv preprint arXiv:2312.00860","author":"Cen Jiazhong","year":"2023","unstructured":"Jiazhong Cen, Jiemin Fang, Chen Yang, Lingxi Xie, Xiaopeng Zhang, Wei Shen, and Qi Tian. 2023. Segment Any 3D Gaussians. arXiv preprint arXiv:2312.00860 (2023)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01840"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19824-3_20"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161534"},{"key":"e_1_3_2_1_6_1","volume-title":"Mvsplat: Efficient 3d gaussian splatting from sparse multi-view images. arXiv preprint arXiv:2403.14627","author":"Chen Yuedong","year":"2024","unstructured":"Yuedong Chen, Haofei Xu, Chuanxia Zheng, Bohan Zhuang, Marc Pollefeys, Andreas Geiger, Tat-Jen Cham, and Jianfei Cai. 2024a. Mvsplat: Efficient 3d gaussian splatting from sparse multi-view images. arXiv preprint arXiv:2403.14627 (2024)."},{"key":"e_1_3_2_1_7_1","volume-title":"MVSplat360: Feed-Forward 360 Scene Synthesis from Sparse Views. arXiv preprint arXiv:2411.04924","author":"Chen Yuedong","year":"2024","unstructured":"Yuedong Chen, Chuanxia Zheng, Haofei Xu, Bohan Zhuang, Andrea Vedaldi, Tat-Jen Cham, and Jianfei Cai. 2024b. MVSplat360: Feed-Forward 360 Scene Synthesis from Sparse Views. arXiv preprint arXiv:2411.04924 (2024)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00127"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00677"},{"key":"e_1_3_2_1_10_1","volume-title":"Cosseggaussians: Compact and swift scene segmenting 3d gaussians. arXiv preprint arXiv:2401.05925","author":"Dou Bin","year":"2024","unstructured":"Bin Dou, Tianyu Zhang, Yongjia Ma, Zhaohui Wang, and Zejian Yuan. 2024. Cosseggaussians: Compact and swift scene segmenting 3d gaussians. arXiv preprint arXiv:2401.05925 (2024)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_31"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00409"},{"key":"e_1_3_2_1_13_1","volume-title":"EgoLifter: Open-world 3D Segmentation for Egocentric Perception. arXiv preprint arXiv:2403.18118","author":"Gu Qiao","year":"2024","unstructured":"Qiao Gu, Zhaoyang Lv, Duncan Frost, Simon Green, Julian Straub, and Chris Sweeney. 2024. EgoLifter: Open-world 3D Segmentation for Egocentric Perception. arXiv preprint arXiv:2403.18118 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"Semantic abstraction: Open-world 3d scene understanding from 2d vision-language models. arXiv preprint arXiv:2207.11514","author":"Ha Huy","year":"2022","unstructured":"Huy Ha and Shuran Song. 2022. Semantic abstraction: Open-world 3d scene understanding from 2d vision-language models. arXiv preprint arXiv:2207.11514 (2022)."},{"key":"e_1_3_2_1_15_1","volume-title":"S4D: Streaming 4D Real-World Reconstruction with Gaussians and 3D Control Points. arXiv preprint arXiv:2408.13036","author":"He Bing","year":"2024","unstructured":"Bing He, Yunuo Chen, Guo Lu, Li Song, and Wenjun Zhang. 2024. S4D: Streaming 4D Real-World Reconstruction with Gaussians and 3D Control Points. arXiv preprint arXiv:2408.13036 (2024)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160969"},{"key":"e_1_3_2_1_18_1","volume-title":"Conceptfusion: Open-set multimodal 3d mapping. arXiv preprint arXiv:2302.07241","author":"Jatavallabhula Krishna Murthy","year":"2023","unstructured":"Krishna Murthy Jatavallabhula, Alihusein Kuwajerwala, Qiao Gu, Mohd Omama, Tao Chen, Alaa Maalouf, Shuang Li, Ganesh Iyer, Soroush Saryazdi, Nikhil Keetha, et al., 2023. Conceptfusion: Open-set multimodal 3d mapping. arXiv preprint arXiv:2302.07241 (2023)."},{"key":"e_1_3_2_1_19_1","volume-title":"FastLGS: Speeding up Language Embedded Gaussians with Feature Grid Mapping. arXiv preprint arXiv:2406.01916","author":"Ji Yuzhou","year":"2024","unstructured":"Yuzhou Ji, He Zhu, Junshu Tang, Wuyi Liu, Zhizhong Zhang, Yuan Xie, Lizhuang Ma, and Xin Tan. 2024. FastLGS: Speeding up Language Embedded Gaussians with Feature Grid Mapping. arXiv preprint arXiv:2406.01916 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01807"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_1_23_1","first-page":"23311","article-title":"Decomposing nerf for editing via feature field distillation","volume":"35","author":"Kobayashi Sosuke","year":"2022","unstructured":"Sosuke Kobayashi, Eiichi Matsumoto, and Vincent Sitzmann. 2022. Decomposing nerf for editing via feature field distillation. Advances in Neural Information Processing Systems, Vol. 35 (2022), 23311-23330.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_24_1","volume-title":"Language-driven semantic segmentation. arXiv preprint arXiv:2201.03546","author":"Li Boyi","year":"2022","unstructured":"Boyi Li, Kilian Q Weinberger, Serge Belongie, Vladlen Koltun, and Ren\u00e9 Ranftl. 2022. Language-driven semantic segmentation. arXiv preprint arXiv:2201.03546 (2022)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00813"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00682"},{"key":"e_1_3_2_1_27_1","volume-title":"ReconX: Reconstruct Any Scene from Sparse Views with Video Diffusion Model. arXiv preprint arXiv:2408.16767","author":"Liu Fangfu","year":"2024","unstructured":"Fangfu Liu, Wenqiang Sun, Hanyang Wang, Yikai Wang, Haowen Sun, Junliang Ye, Jun Zhang, and Yueqi Duan. 2024. ReconX: Reconstruct Any Scene from Sparse Views with Video Diffusion Model. arXiv preprint arXiv:2408.16767 (2024)."},{"key":"e_1_3_2_1_28_1","first-page":"53433","article-title":"Weakly supervised 3d open-vocabulary segmentation","volume":"36","author":"Liu Kunhao","year":"2023","unstructured":"Kunhao Liu, Fangneng Zhan, Jiahui Zhang, Muyu Xu, Yingchen Yu, Abdulmotaleb El Saddik, Christian Theobalt, Eric Xing, and Shijian Lu. 2023. Weakly supervised 3d open-vocabulary segmentation. Advances in Neural Information Processing Systems, Vol. 36 (2023), 53433-53456.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_29_1","volume-title":"Gaga: Group Any Gaussians via 3D-aware Memory Bank. arXiv preprint arXiv:2404.07977","author":"Lyu Weijie","year":"2024","unstructured":"Weijie Lyu, Xueting Li, Abhijit Kundu, Yi-Hsuan Tsai, and Ming-Hsuan Yang. 2024. Gaga: Group Any Gaussians via 3D-aware Memory Bank. arXiv preprint arXiv:2404.07977 (2024)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160800"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00085"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01895"},{"key":"e_1_3_2_1_34_1","volume-title":"International conference on machine learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748-8763."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00510"},{"key":"e_1_3_2_1_36_1","volume-title":"Openmask3d: Open-vocabulary 3d instance segmentation. arXiv preprint arXiv:2306.13631","author":"Takmaz Ay\u00e7a","year":"2023","unstructured":"Ay\u00e7a Takmaz, Elisabetta Fedele, Robert W Sumner, Marc Pollefeys, Federico Tombari, and Francis Engelmann. 2023. Openmask3d: Open-vocabulary 3d instance segmentation. arXiv preprint arXiv:2306.13631 (2023)."},{"key":"e_1_3_2_1_37_1","volume-title":"Dreamgaussian: Generative gaussian splatting for efficient 3d content creation. arXiv preprint arXiv:2309.16653","author":"Tang Jiaxiang","year":"2023","unstructured":"Jiaxiang Tang, Jiawei Ren, Hang Zhou, Ziwei Liu, and Gang Zeng. 2023. Dreamgaussian: Generative gaussian splatting for efficient 3d content creation. arXiv preprint arXiv:2309.16653 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"FreeSplat: Generalizable 3D Gaussian Splatting Towards Free-View Synthesis of Indoor Scenes. arXiv preprint arXiv:2405.17958","author":"Wang Yunsong","year":"2024","unstructured":"Yunsong Wang, Tianxin Huang, Hanlin Chen, and Gim Hee Lee. 2024. FreeSplat: Generalizable 3D Gaussian Splatting Towards Free-View Synthesis of Indoor Scenes. arXiv preprint arXiv:2405.17958 (2024)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01920"},{"key":"e_1_3_2_1_40_1","unstructured":"Yanmin Wu Jiarui Meng Haijie Li Chenming Wu Yahao Shi Xinhua Cheng Chen Zhao Haocheng Feng Errui Ding Jingdong Wang et al. 2024a. OpenGaussian: Towards Point-Level 3D Gaussian-based Open Vocabulary Understanding. arXiv preprint arXiv:2406.02058 (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"Sparsegs: Real-time 360 sparse view synthesis using gaussian splatting. arXiv preprint arXiv:2312.00206","author":"Xiong Haolin","year":"2023","unstructured":"Haolin Xiong, Sairisheek Muttukuru, Rishi Upadhyay, Pradyumna Chari, and Achuta Kadambi. 2023. Sparsegs: Real-time 360 sparse view synthesis using gaussian splatting. arXiv preprint arXiv:2312.00206 (2023)."},{"key":"e_1_3_2_1_42_1","volume-title":"Depthsplat: Connecting gaussian splatting and depth. arXiv preprint arXiv:2410.13862","author":"Xu Haofei","year":"2024","unstructured":"Haofei Xu, Songyou Peng, Fangjinhua Wang, Hermann Blum, Daniel Barath, Andreas Geiger, and Marc Pollefeys. 2024. Depthsplat: Connecting gaussian splatting and depth. arXiv preprint arXiv:2410.13862 (2024)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01922"},{"key":"e_1_3_2_1_44_1","volume-title":"Real-time photorealistic dynamic scene representation and rendering with 4d gaussian splatting. arXiv preprint arXiv:2310.10642","author":"Yang Zeyu","year":"2023","unstructured":"Zeyu Yang, Hongye Yang, Zijie Pan, Xiatian Zhu, and Li Zhang. 2023. Real-time photorealistic dynamic scene representation and rendering with 4d gaussian splatting. arXiv preprint arXiv:2310.10642 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"No Pose","author":"Ye Botao","year":"2024","unstructured":"Botao Ye, Sifei Liu, Haofei Xu, Xueting Li, Marc Pollefeys, Ming-Hsuan Yang, and Songyou Peng. 2024. No Pose, No Problem: Surprisingly Simple 3D Gaussian Splats from Sparse Unposed Images. arXiv preprint arXiv:2410.24207 (2024)."},{"key":"e_1_3_2_1_46_1","volume-title":"Gaussian grouping: Segment and edit anything in 3d scenes. arXiv preprint arXiv:2312.00732","author":"Ye Mingqiao","year":"2023","unstructured":"Mingqiao Ye, Martin Danelljan, Fisher Yu, and Lei Ke. 2023. Gaussian grouping: Segment and edit anything in 3d scenes. arXiv preprint arXiv:2312.00732 (2023)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812291"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00455"},{"key":"e_1_3_2_1_49_1","volume-title":"Transplat: Generalizable 3d gaussian splatting from sparse multi-view images with transformers. arXiv preprint arXiv:2408.13770","author":"Zhang Chuanrui","year":"2024","unstructured":"Chuanrui Zhang, Yingshuang Zou, Zhuoling Li, Minmin Yi, and Haoqian Wang. 2024b. Transplat: Generalizable 3d gaussian splatting from sparse multi-view images with transformers. arXiv preprint arXiv:2408.13770 (2024)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02448"},{"key":"e_1_3_2_1_51_1","volume-title":"Gaussian in the Wild: 3D Gaussian Splatting for Unconstrained Image Collections. arXiv preprint arXiv:2403.15704","author":"Zhang Dongbin","year":"2024","unstructured":"Dongbin Zhang, Chuming Wang, Weitao Wang, Peihao Li, Minghan Qin, and Haoqian Wang. 2024a. Gaussian in the Wild: 3D Gaussian Splatting for Unconstrained Image Collections. arXiv preprint arXiv:2403.15704 (2024)."},{"key":"e_1_3_2_1_52_1","volume-title":"Davison","author":"Zhi Shuaifeng","year":"2021","unstructured":"Shuaifeng Zhi, Tristan Laidlow, Stefan Leutenegger, and Andrew J. Davison. 2021. In-Place Scene Labelling and Understanding with Implicit Scene Representation."},{"key":"e_1_3_2_1_53_1","volume-title":"Stereo magnification: Learning view synthesis using multiplane images. arXiv preprint arXiv:1805.09817","author":"Zhou Tinghui","year":"2018","unstructured":"Tinghui Zhou, Richard Tucker, John Flynn, Graham Fyffe, and Noah Snavely. 2018. Stereo magnification: Learning view synthesis using multiplane images. arXiv preprint arXiv:1805.09817 (2018)."},{"key":"e_1_3_2_1_54_1","volume-title":"Fsgs: Real-time few-shot view synthesis using gaussian splatting. arXiv preprint arXiv:2312.00451","author":"Zhu Zehao","year":"2023","unstructured":"Zehao Zhu, Zhiwen Fan, Yifan Jiang, and Zhangyang Wang. 2023. Fsgs: Real-time few-shot view synthesis using gaussian splatting. arXiv preprint arXiv:2312.00451 (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754964","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:15:49Z","timestamp":1765340149000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754964"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":54,"alternative-id":["10.1145\/3746027.3754964","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754964","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}