{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T12:42:13Z","timestamp":1780576933518,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681517","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"1370-1378","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["3D Question Answering with Scene Graph Reasoning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2103-5037","authenticated-orcid":false,"given":"Zizhao","family":"Wu","sequence":"first","affiliation":[{"name":"Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4630-7344","authenticated-orcid":false,"given":"Haohan","family":"Li","sequence":"additional","affiliation":[{"name":"Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4844-6855","authenticated-orcid":false,"given":"Gongyi","family":"Chen","sequence":"additional","affiliation":[{"name":"Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8407-1137","authenticated-orcid":false,"given":"Zhou","family":"Yu","sequence":"additional","affiliation":[{"name":"Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2876-1771","authenticated-orcid":false,"given":"Xiaoling","family":"Gu","sequence":"additional","affiliation":[{"name":"Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4131-2719","authenticated-orcid":false,"given":"Yigang","family":"Wang","sequence":"additional","affiliation":[{"name":"Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Computer Vision - ECCV 2020 (Lecture Notes in Computer Science","volume":"440","author":"Achlioptas Panos","unstructured":"Panos Achlioptas, Ahmed Abdelreheem, Fei Xia, Mohamed Elhoseiny, and Leonidas J. Guibas. 2020. ReferIt3D: Neural Listeners for Fine-Grained 3D Object Identification in Real-World Scenes. In Computer Vision - ECCV 2020 (Lecture Notes in Computer Science, Vol. 12346), Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). 422--440."},{"key":"e_1_3_2_2_2_1","volume-title":"Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering","author":"Anderson Peter","unstructured":"Peter Anderson, Xiaodong He, Chris Buehler, Damien Teney, Mark Johnson, Stephen Gould, and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. Computer Vision Foundation \/ IEEE Computer Society, 6077--6086."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01854"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1162\/089976603321780317"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01597"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58565-5_13"},{"key":"e_1_3_2_2_7_1","volume-title":"IEEE Conference on Computer Vision and Pattern Recognition, CVPR. Computer Vision Foundation \/ IEEE, 3193--3203","author":"Chen Dave Zhenyu","unstructured":"Dave Zhenyu Chen, Ali Gholami, Matthias Nie\u00dfner, and Angel X. Chang. 2021. Scan2Cap: Context-Aware Dense Captioning in RGB-D Scans. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. Computer Vision Foundation \/ IEEE, 3193--3203."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"e_1_3_2_2_9_1","volume-title":"34th British Machine Vision Conference","author":"Delitzas Alexandros","year":"2023","unstructured":"Alexandros Delitzas, Maria Parelli, Nikolas Hars, Georgios Vlassis, Sotirios- Konstantinos Anagnostidis, Gregor Bachmann, and Thomas Hofmann. 2023. Multi-CLIP: Contrastive Vision-Language Pre-training for Question Answering tasks in 3D Scenes. In 34th British Machine Vision Conference 2023, BMVC. BMVA Press, 748--749."},{"key":"e_1_3_2_2_10_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In the Conference of the North American","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In the Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT, Jill Burstein, Christy Doran, and Thamar Solorio (Eds.). 4171--4186."},{"key":"e_1_3_2_2_11_1","volume-title":"A Generalization of Trans former Networks to Graphs. CoRR abs\/2012.09699","author":"Dwivedi Vijay Prakash","year":"2020","unstructured":"Vijay Prakash Dwivedi and Xavier Bresson. 2020. A Generalization of Trans former Networks to Graphs. CoRR abs\/2012.09699 (2020)."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3104937"},{"key":"e_1_3_2_2_13_1","unstructured":"Wencheng Han Dongqian Guo Cheng-Zhong Xu and Jianbing Shen. 2024. DME-Driver: Integrating Human Decision Logic and 3D Scene Perception in Autonomous Driving. CoRR abs\/2401.03641 (2024)"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_2_15_1","volume-title":"GELU Activation Function in Deep Learning: A Comprehensive Mathematical Analysis and Performance. CoRR abs\/2305.12073","author":"Lee Minhyeok","year":"2023","unstructured":"Minhyeok Lee. 2023. GELU Activation Function in Deep Learning: A Comprehensive Mathematical Analysis and Performance. CoRR abs\/2305.12073 (2023)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_2_2_17_1","volume-title":"GraghVQA: Language-Guided Graph Neural Networks for Graph-based Visual Question Answering. CoRR abs\/2104.10283","author":"Liang Weixin","year":"2021","unstructured":"Weixin Liang, Yanhao Jiang, and Zixuan Liu. 2021. GraghVQA: Language-Guided Graph Neural Networks for Graph-based Visual Question Answering. CoRR abs\/2104.10283 (2021)."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3097435"},{"key":"e_1_3_2_2_19_1","unstructured":"Xiaojian Ma Silong Yong Zilong Zheng Qing Li Yitao Liang Song-Chun Zhu and Siyuan Huang. 2023. SQA3D: Situated Question Answering in 3D Scenes. In ICLR. OpenReview.net."},{"key":"e_1_3_2_2_20_1","unstructured":"Will Norcliffe-Brown Stathis Vafeias and Sarah Parisot. 2018. Learning Conditioned Graph Structures for Interpretable Visual Question Answering. In Advances in Neural Information Processing Systems (NeurIPS) Samy Bengio Hanna M. Wallach Hugo Larochelle Kristen Grauman Nicol\u00f2 Cesa-Bianchi and Roman Garnett (Eds.). 8344--8353."},{"key":"e_1_3_2_2_21_1","volume-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR. 5607--5612","author":"Parelli Maria","unstructured":"Maria Parelli, Alexandros Delitzas, Nikolas Hars, Georgios Vlassis, Sotiris Anag- nostidis, Gregor Bachmann, and Thomas Hofmann. [n. d.]. CLIP-Guided Vision-Language Pre-training for Question Answering in 3D Scenes. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR. 5607--5612."},{"key":"e_1_3_2_2_22_1","volume-title":"Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing, EMNLP, Alessandro Moschitti, Bo Pang, and Walter Daelemans (Eds.). ACL, 1532--1543","author":"Pennington Jeffrey","unstructured":"Jeffrey Pennington, Richard Socher, and Christopher D. Manning. 2014. Glove: Global Vectors for Word Representation. In Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing, EMNLP, Alessandro Moschitti, Bo Pang, and Walter Daelemans (Eds.). ACL, 1532--1543."},{"key":"e_1_3_2_2_23_1","volume-title":"IEEE\/CVF International Conference on Computer Vision, ICCV. IEEE, 9276?9285","author":"Qi Charles R.","unstructured":"Charles R. Qi, Or Litany, Kaiming He, and Leonidas J. Guibas. 2019. Deep Hough Voting for 3D Object Detection in Point Clouds. In IEEE\/CVF International Conference on Computer Vision, ICCV. IEEE, 9276?9285."},{"key":"e_1_3_2_2_24_1","volume-title":"Guibas","author":"Qi Charles Ruizhongtai","year":"2017","unstructured":"Charles Ruizhongtai Qi, Li Yi, Hao Su, and Leonidas J. Guibas. 2017. PointNet++: Deep Hierarchical Feature Learning on Point Sets in a Metric Space. In Advances in Neural Information Processing Systems (NeurIPS), Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna M. Wallach, Rob Fergus, S. V. N. Vishwanathan, and Roman Garnett (Eds.). 5099--5108."},{"key":"e_1_3_2_2_25_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML (Proceedings of Machine Learning Research","volume":"8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, ICML (Proceedings of Machine Learning Research, Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 8748--8763."},{"key":"e_1_3_2_2_26_1","volume-title":"Habitat: A Platform for Embodied AI Research. In IEEE\/CVF International Conference on Computer Vision, ICCV. IEEE, 9338--9346","author":"Savva Manolis","year":"2019","unstructured":"Manolis Savva, Jitendra Malik, Devi Parikh, Dhruv Batra, Abhishek Kadian, Oleksandr Maksymets, Yili Zhao, Erik Wijmans, Bhavana Jain, Julian Straub, Jia Liu, and Vladlen Koltun. 2019. Habitat: A Platform for Embodied AI Research. In IEEE\/CVF International Conference on Computer Vision, ICCV. IEEE, 9338--9346."},{"key":"e_1_3_2_2_27_1","volume-title":"Self- GraphVQA: A Self-Supervised Graph Neural Network for Scene-based Question-Answering. In IEEE\/CVF International Conference on Computer Vision, ICCV - Workshops. IEEE, 4642--4647","author":"Souza Bruno","year":"2023","unstructured":"Bruno Souza, Marius Aasan, H\u00e9lio Pedrini, and Ad\u00edn Ram\u00edrez Rivera. 2023. Self- GraphVQA: A Self-Supervised Graph Neural Network for Scene-based Question-Answering. In IEEE\/CVF International Conference on Computer Vision, ICCV - Workshops. IEEE, 4642--4647."},{"key":"e_1_3_2_2_28_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems (NeurIPS). 5998--6008."},{"key":"e_1_3_2_2_29_1","volume-title":"Graph Attention Networks. In International Conference on Learning Representations, ICLR. OpenReview.net.","author":"Velickovic Petar","year":"2018","unstructured":"Petar Velickovic, Guillem Cucurull, Arantxa Casanova, Adriana Romero, Pietro Li\u00f2, and Yoshua Bengio. 2018. Graph Attention Networks. In International Conference on Learning Representations, ICLR. OpenReview.net."},{"key":"e_1_3_2_2_30_1","volume-title":"William Yang Wang, and Lei Zhang","author":"Wang Xin","year":"2019","unstructured":"Xin Wang, Qiuyuan Huang, Asli Celikyilmaz, Jianfeng Gao, Dinghan Shen, Yuan-Fang Wang, William Yang Wang, and Lei Zhang. 2019. Reinforced Cross-Modal Matching and Self-Supervised Imitation Learning for Vision-Language Navigation. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. Computer Vision Foundation \/ IEEE, 6629--6638."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01973"},{"key":"e_1_3_2_2_32_1","volume-title":"SGEITL: Scene Graph Enhanced Image-Text Learning for Visual Commonsense Reasoning. In Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI. AAAI Press, 5914--5922","author":"Wang Zhecan","year":"2022","unstructured":"Zhecan Wang, Haoxuan You, Liunian Harold Li, Alireza Zareian, Suji Park, Yiqing Liang, Kai-Wei Chang, and Shih-Fu Chang. 2022. SGEITL: Scene Graph Enhanced Image-Text Learning for Visual Commonsense Reasoning. In Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI. AAAI Press, 5914--5922."},{"key":"e_1_3_2_2_33_1","volume-title":"Extending Multi-modal Contrastive Representations. CoRR abs\/2310.08884","author":"Wang Zehan","year":"2023","unstructured":"Zehan Wang, Ziang Zhang, Luping Liu, Yang Zhao, Haifeng Huang, Tao Jin, and Zhou Zhao. 2023. Extending Multi-modal Contrastive Representations. CoRR abs\/2310.08884 (2023)."},{"key":"e_1_3_2_2_34_1","volume-title":"Gibson Env: Real-World Perception for Embodied Agents. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. Computer Vision Foundation \/ IEEE Computer Society, 9068--9079","author":"Xia Fei","year":"2018","unstructured":"Fei Xia, Amir R. Zamir, Zhi-Yang He, Alexander Sax, Jitendra Malik, and Silvio Savarese. 2018. Gibson Env: Real-World Perception for Embodied Agents. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR. Computer Vision Foundation \/ IEEE Computer Society, 9068--9079."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2022.3225327"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3296889"},{"key":"e_1_3_2_2_37_1","volume-title":"Deep Modular Co-Attention Networks for Visual Question Answering. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2019","author":"Yu Zhou","year":"2019","unstructured":"Zhou Yu, Jun Yu, Yuhao Cui, Dacheng Tao, and Qi Tian. 2019. Deep Modular Co-Attention Networks for Visual Question Answering. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2019, Long Beach, CA, USA, June 16-20, 2019. Computer Vision Foundation \/ IEEE, 6281--6290."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00837"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00292"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-020-08790-0"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00272"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681517","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681517","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:48Z","timestamp":1750294668000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681517"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":41,"alternative-id":["10.1145\/3664647.3681517","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681517","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}