{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T09:59:40Z","timestamp":1775815180382,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T00:00:00Z","timestamp":1746662400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62177001"],"award-info":[{"award-number":["62177001"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,8]]},"DOI":"10.1145\/3701716.3717548","type":"proceedings-article","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T16:12:56Z","timestamp":1748016776000},"page":"2314-2319","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["LLM-EvRep: Learning an LLM-Compatible Event Representation Using a Self-Supervised Framework"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-1098-9793","authenticated-orcid":false,"given":"Zongyou","family":"Yu","sequence":"first","affiliation":[{"name":"Beijing Technology and Business University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6648-5050","authenticated-orcid":false,"given":"Qiang","family":"Qu","sequence":"additional","affiliation":[{"name":"The University of Sydney, Sydney, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4824-3158","authenticated-orcid":false,"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[{"name":"Beijing Technology and Business University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4904-7857","authenticated-orcid":false,"given":"Nan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Beijing Technology and Business University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7503-3021","authenticated-orcid":false,"given":"Xiaoming","family":"Chen","sequence":"additional","affiliation":[{"name":"Beijing Technology and Business University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,5,23]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"MiniGPT-v2: Large Language Model as a Unified Interface for Vision-Language Multi-Task Learning. arXiv preprint arXiv:2310.09478","author":"Chen Jun","year":"2023","unstructured":"Jun Chen, Deyao Zhu, Xiaoqian Shen, Xiang Li, Zechun Liu, Pengchuan Zhang, Raghuraman Krishnamoorthi, Vikas Chandra, Yunyang Xiong, and Mohamed Elhoseiny. 2023. MiniGPT-v2: Large Language Model as a Unified Interface for Vision-Language Multi-Task Learning. arXiv preprint arXiv:2310.09478 (2023)."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_2_3_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2004.383"},{"key":"e_1_3_2_2_5_1","volume-title":"Scene-llm: Extending language model for 3d visual understanding and reasoning. arXiv preprint arXiv:2403.11401","author":"Fu Rao","year":"2024","unstructured":"Rao Fu, Jingyu Liu, Xilun Chen, Yixin Nie, and Wenhan Xiong. 2024. Scene-llm: Extending language model for 3d visual understanding and reasoning. arXiv preprint arXiv:2403.11401 (2024)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Guillermo Gallego Tobi Delbr\u00fcck Garrick Orchard Chiara Bartolozzi Brian Taba Andrea Censi Stefan Leutenegger Andrew J Davison J\u00f6rg Conradt Kostas Daniilidis et al. 2020. Event-based vision: A survey. IEEE transactions on pattern analysis and machine intelligence Vol. 44 1 (2020) 154--180.","DOI":"10.1109\/TPAMI.2020.3008413"},{"key":"e_1_3_2_2_7_1","volume-title":"Elias Mueggler, Henri Rebecq, Tobi Delbruck, and Davide Scaramuzza.","author":"Gallego Guillermo","year":"2017","unstructured":"Guillermo Gallego, Jon EA Lund, Elias Mueggler, Henri Rebecq, Tobi Delbruck, and Davide Scaramuzza. 2017. Event-based, 6-DOF camera tracking from photometric depth maps. IEEE transactions on pattern analysis and machine intelligence, Vol. 40, 10 (2017), 2402--2412."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01258-8_46"},{"key":"e_1_3_2_2_9_1","volume-title":"Are high-resolution event cameras really needed? arXiv preprint arXiv:2203.14672","author":"Gehrig Daniel","year":"2022","unstructured":"Daniel Gehrig and Davide Scaramuzza. 2022. Are high-resolution event cameras really needed? arXiv preprint arXiv:2203.14672 (2022)."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-024-07409-w"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01334"},{"key":"e_1_3_2_2_12_1","first-page":"7167","article-title":"Self-supervised learning of event-based optical flow with spiking neural networks","volume":"34","author":"Hagenaars Jesse","year":"2021","unstructured":"Jesse Hagenaars, Federico Paredes-Vall\u00e9s, and Guido De Croon. 2021. Self-supervised learning of event-based optical flow with spiking neural networks. Advances in Neural Information Processing Systems, Vol. 34 (2021), 7167--7179.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00536"},{"key":"e_1_3_2_2_14_1","first-page":"547","article-title":"\u00c9tude comparative de la distribution florale dans une portion des Alpes et des Jura","volume":"37","author":"Jaccard Paul","year":"1901","unstructured":"Paul Jaccard. 1901. \u00c9tude comparative de la distribution florale dans une portion des Alpes et des Jura. Bulletin de la Soci\u00e9t\u00e9 Vaudoise des Sciences Naturelles, Vol. 37 (1901), 547--579.","journal-title":"Bulletin de la Soci\u00e9t\u00e9 Vaudoise des Sciences Naturelles"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3249579"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00215"},{"key":"e_1_3_2_2_17_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01485"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58526-6_22"},{"key":"e_1_3_2_2_20_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li and Yong Jae Lee. 2023. Improved Baselines with Visual Instruction Tuning."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_2_2_22_1","unstructured":"OpenAI. 2024. GPT-4 Technical Report. (2024). https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"e_1_3_2_2_23_1","volume-title":"Converting static image datasets to spiking neuromorphic datasets using saccades. Frontiers in neuroscience","author":"Orchard Garrick","year":"2015","unstructured":"Garrick Orchard, Ajinkya Jayawant, Gregory K Cohen, and Nitish Thakor. 2015. Converting static image datasets to spiking neuromorphic datasets using saccades. Frontiers in neuroscience, Vol. 9 (2015), 437."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28263"},{"key":"e_1_3_2_2_25_1","volume-title":"High speed and high dynamic range video with an event camera","author":"Rebecq Henri","year":"2019","unstructured":"Henri Rebecq, Ren\u00e9 Ranftl, Vladlen Koltun, and Davide Scaramuzza. 2019. High speed and high dynamic range video with an event camera. IEEE transactions on pattern analysis and machine intelligence, Vol. 43, 6 (2019), 1964--1980."},{"key":"e_1_3_2_2_26_1","unstructured":"Irwin Sobel. 1968. A 3x3 Isotropic Gradient Operator for Image Processing. Technical Report Technical Report. Stanford Artificial Intelligence Project (SAIL)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2017.7870263"},{"key":"e_1_3_2_2_28_1","volume-title":"International Conference on Neural Information Processing. Springer, 470--482","author":"Su Menghao","year":"2023","unstructured":"Menghao Su, Panpan Yang, Runhao Jiang, and Rui Yan. 2023. Event-based object recognition using feature fusion and spiking neural networks. In International Conference on Neural Information Processing. Springer, 470--482."},{"key":"e_1_3_2_2_29_1","volume-title":"International conference on machine learning. PMLR, 10096--10106","author":"Tan Mingxing","year":"2021","unstructured":"Mingxing Tan and Quoc Le. 2021. Efficientnetv2: Smaller models and faster training. In International conference on machine learning. PMLR, 10096--10106."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680730"},{"key":"e_1_3_2_2_31_1","volume-title":"Eventclip: Adapting clip for event-based object recognition. arXiv preprint arXiv:2306.06354","author":"Wu Ziyi","year":"2023","unstructured":"Ziyi Wu, Xudong Liu, and Igor Gilitschenski. 2023. Eventclip: Adapting clip for event-based object recognition. arXiv preprint arXiv:2306.06354 (2023)."},{"key":"e_1_3_2_2_32_1","volume-title":"Can Large Language Models Grasp Event Signals? Exploring Pure Zero-Shot Event-based Recognition. arXiv preprint arXiv:2409.09628","author":"Yu Zongyou","year":"2024","unstructured":"Zongyou Yu, Qiang Qu, Xiaoming Chen, and Chen Wang. 2024. Can Large Language Models Grasp Event Signals? Exploring Pure Zero-Shot Event-based Recognition. arXiv preprint arXiv:2409.09628 (2024)."},{"key":"e_1_3_2_2_33_1","volume-title":"International Journal of Computer Vision","author":"Zang Yuhang","year":"2024","unstructured":"Yuhang Zang, Wei Li, Jun Han, Kaiyang Zhou, and Chen Change Loy. 2024. Contextual object detection with multimodal large language models. International Journal of Computer Vision (2024), 1--19."},{"key":"e_1_3_2_2_34_1","volume-title":"Deep learning for event-based vision: A comprehensive survey and benchmarks. arXiv preprint arXiv:2302.08890","author":"Zheng Xu","year":"2023","unstructured":"Xu Zheng, Yexin Liu, Yunfan Lu, Tongyan Hua, Tianbo Pan, Weiming Zhang, Dacheng Tao, and Lin Wang. 2023. Deep learning for event-based vision: A comprehensive survey and benchmarks. arXiv preprint arXiv:2302.08890 (2023)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01652"},{"key":"e_1_3_2_2_36_1","volume-title":"E-clip: Towards label-efficient event-based open-world understanding by clip. arXiv preprint arXiv:2308.03135","author":"Zhou Jiazhou","year":"2023","unstructured":"Jiazhou Zhou, Xu Zheng, Yuanhuiyi Lyu, and Lin Wang. 2023. E-clip: Towards label-efficient event-based open-world understanding by clip. arXiv preprint arXiv:2308.03135 (2023)."},{"key":"e_1_3_2_2_37_1","volume-title":"Eventbind: Learning a unified representation to bind them all for event-based open-world understanding.","author":"Zhou Jiazhou","year":"2024","unstructured":"Jiazhou Zhou, Xu Zheng, Yuanhuiyi Lyu, and Lin Wang. 2024. Eventbind: Learning a unified representation to bind them all for event-based open-world understanding. (2024)."},{"key":"e_1_3_2_2_38_1","volume-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3428487"}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3717548","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3701716.3717548","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T03:01:44Z","timestamp":1759892504000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3717548"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,8]]},"references-count":39,"alternative-id":["10.1145\/3701716.3717548","10.1145\/3701716"],"URL":"https:\/\/doi.org\/10.1145\/3701716.3717548","relation":{},"subject":[],"published":{"date-parts":[[2025,5,8]]},"assertion":[{"value":"2025-05-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}