{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T21:28:22Z","timestamp":1786656502032,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62276103"],"award-info":[{"award-number":["62276103"]}]},{"name":"National Natural Science Foundation of China","award":["62576116"],"award-info":[{"award-number":["62576116"]}]},{"name":"Guangdong-Hong Kong-Macao Center for Applied Mathematics","award":["2025A1515060002"],"award-info":[{"award-number":["2025A1515060002"]}]},{"name":"Guangdong Basic and Applied Basic Research Foundation","award":["2026A1515011836"],"award-info":[{"award-number":["2026A1515011836"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,13]]},"DOI":"10.1145\/3795101.3814657","type":"proceedings-article","created":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T20:45:38Z","timestamp":1786653938000},"page":"1482-1490","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["RE-MMNAS: LLM Reasoning-Guided Evolutionary Architecture Search for Multi-Modal Feature Fusion"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-1525-7042","authenticated-orcid":false,"given":"Haowen","family":"Xiao","sequence":"first","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7809-3436","authenticated-orcid":false,"given":"Xueming","family":"Yan","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Guangzhou, China"},{"name":"Guangdong University of Foreign Studies, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7959-4563","authenticated-orcid":false,"given":"Yue","family":"Xie","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Loughborough, United Kingdom"},{"name":"Loughborough University, Loughborough, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1617-4147","authenticated-orcid":false,"given":"Han","family":"Huang","sequence":"additional","affiliation":[{"name":"School of Software Engineering, Zhuhai, China"},{"name":"Sun Yat-sen University, Zhuhai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,8,13]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Hamid Reza Vaezi Joze, and Vishal M. Patel","author":"Abavisani Mahdi","year":"2019","unstructured":"Mahdi Abavisani, Hamid Reza Vaezi Joze, and Vishal M. Patel. 2019. Improving the Performance of Unimodal Dynamic Hand-Gesture Recognition With Multimodal Training. 000, 8938205 (2019)."},{"key":"e_1_3_2_1_2_1","volume-title":"Manuel Montes-y G\u00f3mez, and Fabio A. Gonz\u00e1lez","author":"Arevalo John","year":"2017","unstructured":"John Arevalo, Thamar Solorio, Manuel Montes-y G\u00f3mez, and Fabio A. Gonz\u00e1lez. 2017. Gated Multimodal Units for Information Fusion. (2017)."},{"key":"e_1_3_2_1_3_1","volume-title":"Action Recognition? A New Model and the Kinetics Dataset","author":"Carreira Joao","year":"2017","unstructured":"Joao Carreira and Andrew Zisserman. 2017. Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset. IEEE (2017)."},{"key":"e_1_3_2_1_4_1","unstructured":"Minghao Chen Kan Wu Bolin Ni Houwen Peng Bei Liu Jianlong Fu Hongyang Chao and Haibin Ling. 2021. Searching the Search Space of Vision Transformer. arXiv:2111.14725 [cs.CV] https:\/\/arxiv.org\/abs\/2111.14725"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171\u20134186."},{"key":"e_1_3_2_1_6_1","volume-title":"Multi-scale features are effective for multi-modal classification: An architecture search viewpoint","author":"Fu Pinhan","year":"2024","unstructured":"Pinhan Fu, Xinyan Liang, Yuhua Qian, Qian Guo, Yayu Zhang, Qin Huang, and Ke Tang. 2024. Multi-scale features are effective for multi-modal classification: An architecture search viewpoint. IEEE Transactions on Circuits and Systems for Video Technology (2024)."},{"key":"e_1_3_2_1_7_1","unstructured":"Ian J Goodfellow David Warde-Farley Mehdi Mirza Aaron Courville and Yoshua Bengio. 2013. Maxout Networks."},{"key":"e_1_3_2_1_8_1","volume-title":"Rishabh Dabral, and Arjun Jain.","author":"Gupta Vikram","year":"2019","unstructured":"Vikram Gupta, Sai Kumar Dwivedi, Rishabh Dabral, and Arjun Jain. 2019. Progression Modelling for Online and Early Gesture Detection. (2019)."},{"key":"e_1_3_2_1_9_1","volume-title":"Forty-second International Conference on Machine Learning.","author":"Ji Zipeng","year":"2022","unstructured":"Zipeng Ji, Guanghui Zhu, Chunfeng Yuan, and Yihua Huang. 2022. RZ-NAS: Enhancing LLM-guided Neural Architecture Search via Reflective Zero-Cost Strategy. In Forty-second International Conference on Machine Learning."},{"key":"e_1_3_2_1_10_1","volume-title":"MMTM: Multimodal Transfer Module for CNN Fusion","author":"Vaezi Joze Hamid Reza","year":"2020","unstructured":"Hamid Reza Vaezi Joze, Amirreza Shaban, Michael L. Iuzzolino, and Kazuhito Koishida. 2020. MMTM: Multimodal Transfer Module for CNN Fusion. IEEE (2020)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i12.29281"},{"key":"e_1_3_2_1_12_1","first-page":"1","article-title":"Evolutionary Deep Fusion Method and Its Application in Chemical Structure Recognition","volume":"99","author":"Liang Xinyan","year":"2021","unstructured":"Xinyan Liang, Qian Guo, Yuhua Qian, Weiping Ding, and Qingfu Zhang. 2021. Evolutionary Deep Fusion Method and Its Application in Chemical Structure Recognition. IEEE Transactions on Evolutionary Computation PP, 99 (2021), 1\u20131.","journal-title":"IEEE Transactions on Evolutionary Computation PP"},{"key":"e_1_3_2_1_13_1","volume-title":"Darts: Differentiable architecture search. arXiv preprint arXiv:1806.09055","author":"Liu Hanxiao","year":"2018","unstructured":"Hanxiao Liu, Karen Simonyan, and Yiming Yang. 2018. Darts: Differentiable architecture search. arXiv preprint arXiv:1806.09055 (2018)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1209"},{"key":"e_1_3_2_1_15_1","volume-title":"A multiscale neural architecture search framework for multimodal fusion. Information science 679, 000","author":"Lv Jindi","year":"2024","unstructured":"Jindi Lv, Yanan Sun, Qing Ye, Wentao Feng, and Jiancheng Lv. 2024. A multiscale neural architecture search framework for multimodal fusion. Information science 679, 000 (2024), 16."},{"key":"e_1_3_2_1_16_1","volume-title":"Shih Yao Lin, and Xiaohui Xie","author":"Ma Haoyu","year":"2021","unstructured":"Haoyu Ma, Liangjian Chen, Deying Kong, Zhe Wang, Xingwei Liu, Hao Tang, Xiangyi Yan, Yusheng Xie, Shih Yao Lin, and Xiaohui Xie. 2021. TransFusion: Cross-view Fusion with Transformer for 3D Human Pose Estimation. (2021)."},{"key":"e_1_3_2_1_17_1","unstructured":"Xiaoyu Ma and Hao Chen. 2025. Revisit Modality Imbalance at the Decision Layer. arXiv:2510.14411 [cs.LG] https:\/\/arxiv.org\/abs\/2510.14411"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.456"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3638529.3654017"},{"key":"e_1_3_2_1_20_1","volume-title":"MFAS: Multimodal Fusion Architecture Search","author":"Perez-Rua Juan Manuel","year":"2019","unstructured":"Juan Manuel Perez-Rua, Valentin Vielzeuf, Stephane Pateux, Moez Baccouche, and Frederic Jurie. 2019. MFAS: Multimodal Fusion Architecture Search. IEEE (2019)."},{"key":"e_1_3_2_1_21_1","volume-title":"Lemo-nade: Multiparameter neural architecture discovery with llms. arXiv preprint arXiv:2402.18443","author":"Rahman Md Hafizur","year":"2024","unstructured":"Md Hafizur Rahman and Prabuddha Chakraborty. 2024. Lemo-nade: Multiparameter neural architecture discovery with llms. arXiv preprint arXiv:2402.18443 (2024)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014780"},{"key":"e_1_3_2_1_23_1","volume-title":"Tian Tsong Ng, and Gang Wang","author":"Shahroudy Amir","year":"2016","unstructured":"Amir Shahroudy, Jun Liu, Tian Tsong Ng, and Gang Wang. 2016. NTU RGB+D: A Large Scale Dataset for 3D Human Activity Analysis. IEEE Computer Society (2016), 1010\u20131019."},{"key":"e_1_3_2_1_24_1","volume-title":"Two-Stream Convolutional Networks for Action Recognition in Videos. Advances in neural information processing systems 1","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-Stream Convolutional Networks for Action Recognition in Videos. Advances in neural information processing systems 1 (2014)."},{"key":"e_1_3_2_1_25_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. Computer Science","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. Computer Science (2014)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Valentin Vielzeuf Alexis Lechervy St\u00e9phane Pateux and Fr\u00e9d\u00e9ric Jurie. 2018. CentralNet: a Multilayer Approach for Multimodal Fusion. (2018).","DOI":"10.1007\/978-3-030-11024-6_44"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IRI51335.2021.00034"},{"key":"e_1_3_2_1_29_1","volume-title":"Deep multimodal fusion by channel exchanging. Advances in neural information processing systems 33","author":"Wang Yikai","year":"2020","unstructured":"Yikai Wang, Wenbing Huang, Fuchun Sun, Tingyang Xu, Yu Rong, and Junzhou Huang. 2020. Deep multimodal fusion by channel exchanging. Advances in neural information processing systems 33 (2020), 4835\u20134845."},{"key":"e_1_3_2_1_30_1","volume-title":"Chi, Quoc Le, and Denny Zhou","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Ed Chi, Quoc Le, and Denny Zhou. 2022. Chain of Thought Prompting Elicits Reasoning in Large Language Models. (2022)."},{"key":"e_1_3_2_1_31_1","volume-title":"Aggregated Residual Transformations for Deep Neural Networks","author":"Xie Saining","year":"2016","unstructured":"Saining Xie, Ross Girshick, Piotr Doll\u00e1r, Zhuowen Tu, and Kaiming He. 2016. Aggregated Residual Transformations for Deep Neural Networks. IEEE (2016)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3193569"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2023.3301774"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3514708"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.108"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20872"},{"key":"e_1_3_2_1_37_1","volume-title":"GPT-NAS: Evolutionary neural architecture search with the generative pre-trained model. arXiv preprint arXiv:2305.05351","author":"Yu Caiyang","year":"2023","unstructured":"Caiyang Yu, Xianggen Liu, Yifan Wang, Yun Liu, Wentao Feng, Xiong Deng, Chenwei Tang, and Jiancheng Lv. 2023. GPT-NAS: Evolutionary neural architecture search with the generative pre-trained model. arXiv preprint arXiv:2305.05351 (2023)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/tip.2021.3087348"},{"key":"e_1_3_2_1_39_1","unstructured":"Zitong Yu Benjia Zhou Jun Wan Pichao Wang and Guoying Zhao. 2020. Searching Multi-Rate and Multi-Modal Temporal Enhanced Networks for Gesture Recognition. (2020)."},{"key":"e_1_3_2_1_40_1","volume-title":"Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250","author":"Zadeh Amir","year":"2017","unstructured":"Amir Zadeh, Minghai Chen, Soujanya Poria, Erik Cambria, and Louis-Philippe Morency. 2017. Tensor fusion network for multimodal sentiment analysis. arXiv preprint arXiv:1707.07250 (2017)."},{"key":"e_1_3_2_1_41_1","volume-title":"Can gpt-4 perform neural architecture search? arXiv preprint arXiv:2304.10970","author":"Zheng Mingkai","year":"2023","unstructured":"Mingkai Zheng, Xiu Su, Shan You, Fei Wang, Chen Qian, Chang Xu, and Samuel Albanie. 2023. Can gpt-4 perform neural architecture search? arXiv preprint arXiv:2304.10970 (2023)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3113183"},{"key":"e_1_3_2_1_43_1","volume-title":"Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578","author":"Zoph Barret","year":"2016","unstructured":"Barret Zoph and Quoc V Le. 2016. Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578 (2016)."}],"event":{"name":"GECCO '26 Companion: Genetic and Evolutionary Computation Conference Companion","location":"Centro Internacional de Convenciones CIC-ANDE San Jose Costa Rica","acronym":"GECCO '26 Companion","sponsor":["SIGEVO ACM Special Interest Group on Genetic and Evolutionary Computation"]},"container-title":["Proceedings of the Genetic and Evolutionary Computation Conference Companion"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3795101.3814657","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T20:47:17Z","timestamp":1786654037000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3795101.3814657"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"references-count":43,"alternative-id":["10.1145\/3795101.3814657","10.1145\/3795101"],"URL":"https:\/\/doi.org\/10.1145\/3795101.3814657","relation":{},"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"2026-08-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}