{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:15:57Z","timestamp":1777655757601,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976158,62376198,62076182"],"award-info":[{"award-number":["61976158,62376198,62076182"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681229","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"4014-4023","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Lite-Mind: Towards Efficient and Robust Brain Representation Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-8961-5437","authenticated-orcid":false,"given":"Zixuan","family":"Gong","sequence":"first","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1037-1361","authenticated-orcid":false,"given":"Qi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0072-6323","authenticated-orcid":false,"given":"Guangyin","family":"Bao","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2993-7142","authenticated-orcid":false,"given":"Lei","family":"Zhu","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6938-204X","authenticated-orcid":false,"given":"Yu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1813-8898","authenticated-orcid":false,"given":"Ke","family":"Liu","sequence":"additional","affiliation":[{"name":"Beijing Anding Hospital, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8588-2177","authenticated-orcid":false,"given":"Liang","family":"Hu","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6588-1468","authenticated-orcid":false,"given":"Duoqian","family":"Miao","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Emily J Allen Ghislain St-Yves Yihan Wu Jesse L Breedlove Jacob S Prince Logan T Dowdle Matthias Nau Brad Caron Franco Pestilli Ian Charest et al. 2022. A massive 7T fMRI dataset to bridge cognitive neuroscience and artificial intelligence. Nature neuroscience 25 1 (2022) 116--126.","DOI":"10.1038\/s41593-021-00962-x"},{"key":"e_1_3_2_1_2_1","unstructured":"Romain Beaumont. 2022. Clip retrieval: Easily compute clip embeddings and build a clip retrieval system with them."},{"key":"e_1_3_2_1_3_1","unstructured":"Defu Cao Yujing Wang Juanyong Duan Ce Zhang Xia Zhu Congrui Huang Yunhai Tong Bixiong Xu Jing Bai Jie Tong et al. 2020. Spectral temporal graph neural network for multivariate time-series forecasting. Advances in neural information processing systems 33 (2020) 17766--17778."},{"key":"e_1_3_2_1_4_1","volume-title":"Unsupervised learning of visual features by contrasting cluster assignments. Advances in neural information processing systems 33","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020. Unsupervised learning of visual features by contrasting cluster assignments. Advances in neural information processing systems 33 (2020), 9912--9924."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02175"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3263181"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00412"},{"key":"e_1_3_2_1_9_1","volume-title":"Decoding natural image stimuli from fMRI data with a surface-based convolutional network. arXiv preprint arXiv:2212.02409","author":"Gu Zijin","year":"2022","unstructured":"Zijin Gu, Keith Jamison, Amy Kuceyeski, and Mert Sabuncu. 2022. Decoding natural image stimuli from fMRI data with a surface-based convolutional network. arXiv preprint arXiv:2212.02409 (2022)."},{"key":"e_1_3_2_1_10_1","volume-title":"Adaptive fourier neural operators: Efficient token mixers for transformers. arXiv preprint arXiv:2111.13587","author":"Guibas John","year":"2021","unstructured":"John Guibas, Morteza Mardani, Zongyi Li, Andrew Tao, Anima Anandkumar, and Bryan Catanzaro. 2021. Adaptive fourier neural operators: Efficient token mixers for transformers. arXiv preprint arXiv:2111.13587 (2021)."},{"key":"e_1_3_2_1_11_1","volume-title":"Generic decoding of seen and imagined objects using hierarchical visual features. Nature communications 8, 1","author":"Horikawa Tomoyasu","year":"2017","unstructured":"Tomoyasu Horikawa and Yukiyasu Kamitani. 2017. Generic decoding of seen and imagined objects using hierarchical visual features. Nature communications 8, 1 (2017), 15037."},{"key":"e_1_3_2_1_12_1","volume-title":"Decoding the visual and subjective contents of the human brain. Nature neuroscience 8, 5","author":"Kamitani Yukiyasu","year":"2005","unstructured":"Yukiyasu Kamitani and Frank Tong. 2005. Decoding the visual and subjective contents of the human brain. Nature neuroscience 8, 5 (2005), 679--685."},{"key":"e_1_3_2_1_13_1","volume-title":"Akilles Rechardt, Jeremy I Skipper, and Gabriella Vigliocco.","author":"Kewenig Viktor","year":"2023","unstructured":"Viktor Kewenig, Christopher Edwards, Quitterie Lacome DEstalenx, Akilles Rechardt, Jeremy I Skipper, and Gabriella Vigliocco. 2023. Evidence of Human-Like Visual-Linguistic Integration in Multimodal Large Language Models During Predictive Language Processing. arXiv preprint arXiv:2308.06035 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"Brain-optimized inference improves reconstructions of fMRI brain activity. arXiv preprint arXiv:2312.07705","author":"Kneeland Reese","year":"2023","unstructured":"Reese Kneeland, Jordyn Ojeda, Ghislain St-Yves, and Thomas Naselaris. 2023. Brain-optimized inference improves reconstructions of fMRI brain activity. arXiv preprint arXiv:2312.07705 (2023)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2022.3228131"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.5555\/3546258.3546299"},{"key":"e_1_3_2_1_17_1","volume-title":"Frequency Spectrum Is More Effective for Multimodal Representation and Fusion: A Multimodal Spectrum Rumor Detector. In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI","author":"Lao An","year":"2024","unstructured":"An Lao, Qi Zhang, Chongyang Shi, Longbing Cao, Kun Yi, Liang Hu, and Duoqian Miao. 2024. Frequency Spectrum Is More Effective for Multimodal Representation and Fusion: A Multimodal Spectrum Rumor Detector. In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI, February 20-27, 2024, Vancouver, Canada. AAAI Press, 18426--18434."},{"key":"e_1_3_2_1_18_1","volume-title":"Fnet: Mixing tokens with fourier transforms. arXiv preprint arXiv:2105.03824","author":"Lee-Thorp James","year":"2021","unstructured":"James Lee-Thorp, Joshua Ainslie, Ilya Eckstein, and Santiago Ontanon. 2021. Fnet: Mixing tokens with fourier transforms. arXiv preprint arXiv:2105.03824 (2021)."},{"key":"e_1_3_2_1_19_1","first-page":"29624","article-title":"Mind reader: Reconstructing complex images from brain activities","volume":"35","author":"Lin Sikun","year":"2022","unstructured":"Sikun Lin, Thomas Sprague, and Ambuj K Singh. 2022. Mind reader: Reconstructing complex images from brain activities. Advances in Neural Information Processing Systems 35 (2022), 29624--29636.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/B978-0-12-822421-2.00008-9"},{"key":"e_1_3_2_1_21_1","volume-title":"BrainCLIP: Bridging Brain and Visual-Linguistic Representation via CLIP for Generic Natural Visual Stimulus Decoding from fMRI. arXiv preprint arXiv:2302.12971","author":"Liu Yulong","year":"2023","unstructured":"Yulong Liu, Yongqiang Ma, Wei Zhou, Guibo Zhu, and Nanning Zheng. 2023. BrainCLIP: Bridging Brain and Visual-Linguistic Representation via CLIP for Generic Natural Visual Stimulus Decoding from fMRI. arXiv preprint arXiv:2302.12971 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."},{"key":"e_1_3_2_1_23_1","volume-title":"Unibrain: Unify image reconstruction and captioning all in one diffusion model from human brain activity. arXiv preprint arXiv:2308.07428","author":"Mai Weijian","year":"2023","unstructured":"Weijian Mai and Zhijun Zhang. 2023. Unibrain: Unify image reconstruction and captioning all in one diffusion model from human brain activity. arXiv preprint arXiv:2308.07428 (2023)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuroimage.2010.07.073"},{"key":"e_1_3_2_1_25_1","volume-title":"Brain-diffuser: Natural scene reconstruction from fmri signals using generative latent diffusion. arXiv preprint arXiv:2303.05334","author":"Ozcelik Furkan","year":"2023","unstructured":"Furkan Ozcelik and Rufin VanRullen. 2023. Brain-diffuser: Natural scene reconstruction from fmri signals using generative latent diffusion. arXiv preprint arXiv:2303.05334 (2023)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2995909"},{"key":"e_1_3_2_1_27_1","volume-title":"Toward a universal decoder of linguistic meaning from brain activation. Nature communications 9, 1","author":"Pereira Francisco","year":"2018","unstructured":"Francisco Pereira, Bin Lou, Brianna Pritchett, Samuel Ritter, Samuel J Gershman, Nancy Kanwisher, Matthew Botvinick, and Evelina Fedorenko. 2018. Toward a universal decoder of linguistic meaning from brain activation. Nature communications 9, 1 (2018), 963."},{"key":"e_1_3_2_1_28_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_29_1","volume-title":"Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)."},{"key":"e_1_3_2_1_30_1","volume-title":"Global filter networks for image classification. Advances in neural information processing systems 34","author":"Rao Yongming","year":"2021","unstructured":"Yongming Rao, Wenliang Zhao, Zheng Zhu, Jiwen Lu, and Jie Zhou. 2021. Global filter networks for image classification. Advances in neural information processing systems 34 (2021), 980--993."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00844"},{"key":"e_1_3_2_1_32_1","first-page":"25278","article-title":"Laion-5b: An open large-scale dataset for training next generation image-text models","volume":"35","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, et al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems 35 (2022), 25278--25294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_33_1","unstructured":"Paul S Scotti Atmadeep Banerjee Jimmie Goode Stepan Shabalin Alex Nguyen Ethan Cohen Aidan J Dempster Nathalie Verlinde Elad Yundler David Weisberg et al. 2023. Reconstructing the Mind's Eye: fMRI-to-Image with Contrastive Learning and Diffusion Priors. arXiv preprint arXiv:2305.18274 (2023)."},{"key":"e_1_3_2_1_34_1","volume-title":"Deep image reconstruction from human brain activity. PLoS computational biology 15, 1","author":"Shen Guohua","year":"2019","unstructured":"Guohua Shen, Tomoyasu Horikawa, Kei Majima, and Yukiyasu Kamitani. 2019. Deep image reconstruction from human brain activity. PLoS computational biology 15, 1 (2019), e1006633."},{"key":"e_1_3_2_1_35_1","unstructured":"Yuge Shi Brooks Paige Philip Torr et al. 2019. Variational mixture-of-experts autoencoders for multi-modal deep generative models. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01389"},{"key":"e_1_3_2_1_37_1","volume-title":"September","author":"Tan Mingxing","year":"2020","unstructured":"Mingxing Tan and V Le Quoc. [n. d.]. EfficientNet: Rethinking Model Scaling for Convolutional Neural Networks, September 2020. arXiv preprint arXiv:1905.11946 ([n. d.])."},{"key":"e_1_3_2_1_38_1","volume-title":"Image quality assessment: from error visibility to structural similarity","author":"Wang Zhou","year":"2004","unstructured":"Zhou Wang, Alan C Bovik, Hamid R Sheikh, and Eero P Simoncelli. 2004. Image quality assessment: from error visibility to structural similarity. IEEE transactions on image processing 13, 4 (2004), 600--612."},{"key":"e_1_3_2_1_39_1","volume-title":"Multimodal generative models for scalable weakly-supervised learning. Advances in neural information processing systems 31","author":"Wu Mike","year":"2018","unstructured":"Mike Wu and Noah Goodman. 2018. Multimodal generative models for scalable weakly-supervised learning. Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_1_40_1","volume-title":"HyDiscGAN: A Hybrid Distributed cGAN for Audio-Visual Privacy Preservation in Multimodal Sentiment Analysis. arXiv preprint arXiv:2404.11938","author":"Wu Zhuojia","year":"2024","unstructured":"Zhuojia Wu, Qi Zhang, Duoqian Miao, Kun Yi, Wei Fan, and Liang Hu. 2024. HyDiscGAN: A Hybrid Distributed cGAN for Audio-Visual Privacy Preservation in Multimodal Sentiment Analysis. arXiv preprint arXiv:2404.11938 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00804"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00181"},{"key":"e_1_3_2_1_43_1","volume-title":"International Conference on Machine Learning. PMLR, 25038--25054","author":"Yang Ling","year":"2022","unstructured":"Ling Yang and Shenda Hong. 2022. Unsupervised time-series representation learning with iterative bilinear temporal-spectral fusion. In International Conference on Machine Learning. PMLR, 25038--25054."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00414"},{"key":"e_1_3_2_1_45_1","unstructured":"Kun Yi Qi Zhang Longbing Cao Shoujin Wang Guodong Long Liang Hu Hui He Zhendong Niu Wei Fan and Hui Xiong. 2023. A Survey on Deep Learning based Time Series Analysis with Frequency Transformation. arXiv:2302.02173 [cs.LG]"},{"key":"e_1_3_2_1_46_1","volume-title":"Advances in Neural Information Processing Systems 36","author":"Yi Kun","year":"2024","unstructured":"Kun Yi, Qi Zhang, Wei Fan, Hui He, Liang Hu, Pengyang Wang, Ning An, Longbing Cao, and Zhendong Niu. 2023. FourierGNN: Rethinking Multivariate Time Series Forecasting from a Pure Graph Perspective. In Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"Frequency-domain MLPs are more effective learners in time series forecasting. Advances in Neural Information Processing Systems 36","author":"Yi Kun","year":"2024","unstructured":"Kun Yi, Qi Zhang, Wei Fan, Shoujin Wang, Pengyang Wang, Hui He, Ning An, Defu Lian, Longbing Cao, and Zhendong Niu. 2024. Frequency-domain MLPs are more effective learners in time series forecasting. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_48_1","volume-title":"MLIP: Efficient Multi-Perspective Language-Image Pretraining with Exhaustive Data Utilization. arXiv preprint arXiv:2406.01460","author":"Zhang Yu","year":"2024","unstructured":"Yu Zhang, Qi Zhang, Zixuan Gong, Yiwei Shi, Yepeng Liu, Duoqian Miao, Yang Liu, Ke Liu, Kun Yi, Wei Fan, et al. 2024. MLIP: Efficient Multi-Perspective Language-Image Pretraining with Exhaustive Data Utilization. arXiv preprint arXiv:2406.01460 (2024)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.54"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681229","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681229","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:03Z","timestamp":1750295883000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681229"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":49,"alternative-id":["10.1145\/3664647.3681229","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681229","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}