{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:26:00Z","timestamp":1783610760193,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62306171"],"award-info":[{"award-number":["62306171"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62136005"],"award-info":[{"award-number":["62136005"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681351","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"9126-9135","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["CoMO-NAS: Core-Structures-Guided Multi-Objective Neural Architecture Search for Multi-Modal Classification"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7647-9921","authenticated-orcid":false,"given":"Pinhan","family":"Fu","sequence":"first","affiliation":[{"name":"Institute of Big Data Science and Industry, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2589-5392","authenticated-orcid":false,"given":"Xinyan","family":"Liang","sequence":"additional","affiliation":[{"name":"Institute of Big Data Science and Industry, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6772-4247","authenticated-orcid":false,"given":"Yuhua","family":"Qian","sequence":"additional","affiliation":[{"name":"Institute of Big Data Science and Industry, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9189-7834","authenticated-orcid":false,"given":"Qian","family":"Guo","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Taiyuan University of Science and Technolog, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8065-1038","authenticated-orcid":false,"given":"Zhifang","family":"Wei","sequence":"additional","affiliation":[{"name":"Institute of Big Data Science and Industry, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8346-2926","authenticated-orcid":false,"given":"Wen","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Big Data Science and Industry, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Manuel Montes-y G\u00f3mez, and FabioA. Gonz\u00e1lez","author":"Arevalo John","year":"2017","unstructured":"John Arevalo, Thamar Solorio, Manuel Montes-y G\u00f3mez, and FabioA. Gonz\u00e1lez. 2017. Gated Multimodal Units for Information Fusion. Cornell University - arXiv (2017)."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 469--478","author":"Baradel Fabien","unstructured":"Fabien Baradel, Christian Wolf, Julien Mille, and Graham W. Taylor. 2018. Glimpse Clouds: Human Activity Recognition from Unstructured Feature Points. In Proceedings of the 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 469--478."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Michael Emmerich Nicola Beume and Boris Naujoks. 2005. An EMO Algorithm Using the Hypervolume Measure as Selection Criterion. 62--76.","DOI":"10.1007\/978-3-540-31880-4_5"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/440"},{"key":"e_1_3_2_1_6_1","volume-title":"Maxout Networks. In Proceedings of the 30th International Conference on International Conference on Machine Learning","volume":"28","author":"Goodfellow Ian J.","year":"2013","unstructured":"Ian J. Goodfellow, David Warde-Farley, Mehdi Mirza, Aaron Courville, and Yoshua Bengio. 2013. Maxout Networks. In Proceedings of the 30th International Conference on International Conference on Machine Learning, Vol. 28. 1319--1327."},{"key":"e_1_3_2_1_7_1","volume-title":"Progression Modelling for Online and Early Gesture Detection. In 2019 International Conference on 3D Vision. 289--297","author":"Gupta Vikram","year":"2019","unstructured":"Vikram Gupta, Sai Kumar Dwivedi, Rishabh Dabral, and Arjun Jain. 2019. Progression Modelling for Online and Early Gesture Detection. In 2019 International Conference on 3D Vision. 289--297."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3171983"},{"key":"e_1_3_2_1_9_1","unstructured":"Ming Hou Jiajia Tang Jianhai Zhang Wanzeng Kong and Qibin Zhao. 2019. Deep Multimodal Multilinear Fusion with High-Order Polynomial Pooling."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3194957"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2022.08.017"},{"key":"e_1_3_2_1_12_1","volume-title":"Hadamard Product for Low-rank Bilinear Pooling. International Conference on Learning Representations","author":"Kim Jin-Hwa","year":"2017","unstructured":"Jin-Hwa Kim, KyoungWoon On, Woosang Lim, Jeong-Hee Kim, Jung-Woo Ha, and Byoung-Tak Zhang. 2017. Hadamard Product for Low-rank Bilinear Pooling. International Conference on Learning Representations (2017), 1--10."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2019.8756576"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.5555\/3304415.3304527"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i12.29281"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2021.3064943"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3125995"},{"key":"e_1_3_2_1_18_1","first-page":"1653","article-title":"Multi-granulation fusion-driven method for many-view classification","volume":"59","author":"Liang Xinyan","year":"2022","unstructured":"Xinyan Liang, Yuhua Qian, Qian Guo, and Qin Huang. 2022. Multi-granulation fusion-driven method for many-view classification. Journal of Computer Research and Development, Vol. 59, 8 (2022), 1653--1667.","journal-title":"Journal of Computer Research and Development"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the International Conference on Learning Representations. 1--11","author":"Liu Hanxiao","year":"2019","unstructured":"Hanxiao Liu, Karen Simonyan, and Yiming Yang. 2019. DARTS: Differentiable Architecture Search. In Proceedings of the International Conference on Learning Representations. 1--11."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i7.26066"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20724"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1209"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.456"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612555"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00806"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00713"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition. 1010--1019","author":"Shahroudy Amir","year":"2016","unstructured":"Amir Shahroudy, Jun Liu, Tian-Tsong Ng, and Gang Wang. 2016. NTU RGBD: A Large Scale Dataset for 3D Human Activity Analysis. In Proceedings of the 2016 IEEE Conference on Computer Vision and Pattern Recognition. 1010--1019."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 27th International Conference on Neural Information Processing Systems","volume":"1","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-Stream Convolutional Networks for Action Recognition in Videos. In Proceedings of the 27th International Conference on Neural Information Processing Systems, Vol. 1. 568--576."},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the Third International Conference on Learning Representations. 1--14","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In Proceedings of the Third International Conference on Learning Representations. 1--14."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2608882"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01330"},{"key":"e_1_3_2_1_32_1","volume-title":"CentralNet: A Multilayer Approach for Multimodal Fusion. In European Conference on Computer Vision Workshops. 575--589","author":"Vielzeuf Valentin","year":"2019","unstructured":"Valentin Vielzeuf, Alexis Lechervy, St\u00e9phane Pateux, and Fr\u00e9d\u00e9ric Jurie. 2019. CentralNet: A Multilayer Approach for Multimodal Fusion. In European Conference on Computer Vision Workshops. 575--589."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612545"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2022.3192635"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i14.29546"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3243521"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00256"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2018.11.015"},{"key":"e_1_3_2_1_39_1","volume-title":"Super Normal Vector for Activity Recognition Using Depth Sequences. In 2014 IEEE Conference on Computer Vision and Pattern Recognition. 804--811","author":"Yang Xiaodong","year":"2014","unstructured":"Xiaodong Yang and YingLi Tian. 2014. Super Normal Vector for Activity Recognition Using Depth Sequences. In 2014 IEEE Conference on Computer Vision and Pattern Recognition. 804--811."},{"key":"e_1_3_2_1_40_1","volume-title":"BM-NAS: Bilevel Multimodal Neural Architecture Search","author":"Yin Yihang","unstructured":"Yihang Yin, Siyu Huang, Xiang Zhang, and Dejing Dou. 2022. BM-NAS: Bilevel Multimodal Neural Architecture Search. In Association for the Advancement of Artificial Intelligence. 8901--8909."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2018.2817340"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3087348"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1115"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3297410"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2022.3140855"},{"key":"e_1_3_2_1_46_1","volume-title":"EgoGesture: A New Dataset and Benchmark for Egocentric Hand Gesture Recognition","author":"Zhang Yifan","year":"2018","unstructured":"Yifan Zhang, Congqi Cao, Jian Cheng, and Hanqing Lu. 2018. EgoGesture: A New Dataset and Benchmark for Egocentric Hand Gesture Recognition. IEEE Transactions on Multimedia (2018), 1038--1050."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00572"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681351","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681351","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:44Z","timestamp":1750295864000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681351"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":47,"alternative-id":["10.1145\/3664647.3681351","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681351","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}