{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:08:45Z","timestamp":1784138925713,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"NSFC","award":["62377028, 62276114, 62477016"],"award-info":[{"award-number":["62377028, 62276114, 62477016"]}]},{"name":"Natural Science Foundation of Guangdong","award":["2024A1515140144"],"award-info":[{"award-number":["2024A1515140144"]}]},{"name":"Guangzhou Science and Technology Planning Project","award":["2025A03J3565, 2023GH01"],"award-info":[{"award-number":["2025A03J3565, 2023GH01"]}]},{"name":"&#x5c;&quot;Master Mentor Plan&#x5c;&quot; of Jinan University","award":["YDXS2501"],"award-info":[{"award-number":["YDXS2501"]}]},{"name":"the Fundamental Research Funds for the Central Universities","award":["21625102, 2026KQZX108"],"award-info":[{"award-number":["21625102, 2026KQZX108"]}]},{"name":"the teaching reform research projects of Jinan University","award":["JG2026030"],"award-info":[{"award-number":["JG2026030"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809586","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"1221-1231","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Mitigating Evidence Suppression: Bi-level Active Evidence Injection for Educational Video Understanding"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9915-4346","authenticated-orcid":false,"given":"Cheng","family":"Liu","sequence":"first","affiliation":[{"name":"Jinan University, guangzhou, Guandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1877-5471","authenticated-orcid":false,"given":"Yiping","family":"Wang","sequence":"additional","affiliation":[{"name":"Jinan University, guangzhou, Guandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6911-3853","authenticated-orcid":false,"given":"Quanlong","family":"Guan","sequence":"additional","affiliation":[{"name":"Jinan University, Guangzhou, Guandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6651-1175","authenticated-orcid":false,"given":"Chaobo","family":"He","sequence":"additional","affiliation":[{"name":"South China Normal University, Guangzhou, Guandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7310-3653","authenticated-orcid":false,"given":"Xingyu","family":"Zhu","sequence":"additional","affiliation":[{"name":"Jinan University, Guangzhou, Guandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6435-6570","authenticated-orcid":false,"given":"Liangda","family":"Fang","sequence":"additional","affiliation":[{"name":"Jinan University, Guangzhou, Guandong, China and Pazhou Laboratory, Guangzhou, Guandong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_1_2_1","volume-title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding. In International Conference on Learning Representations (ICLR).","author":"Chen Guo","year":"2025","unstructured":"Guo Chen, Yicheng Liu, Yifei Huang, Baoqi Pei, Jilan Xu, Yuping He, Tong Lu, Yali Wang, and Limin Wang. 2025. CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_3_1","unstructured":"Zhe Chen Weiyun Wang Yue Cao Yangzhou Liu Zhangwei Gao Erfei Cui Jinguo Zhu Shenglong Ye Hao Tian Zhaoyang Liu et al. 2024a. Expanding Performance Boundaries of Open-Source Multimodal Models with Model Data and Test-Time Scaling. arXiv preprint arXiv:2412.05271 (2024)."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024b. InternVL: Scaling Up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 24185-24198."},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the 2019 ACL Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP. 276-286","author":"Clark Kevin","unstructured":"Kevin Clark, Urvashi Khandelwal, Omer Levy, and Christopher D. Manning. 2019. What Does BERT Look at? An Analysis of BERT's Attention. In Proceedings of the 2019 ACL Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP. 276-286."},{"key":"e_1_3_2_1_6_1","unstructured":"Shizhan Gong Yankai Jiang Qi Dou and Farzan Farnia. 2025. Kernel-based Unsupervised Embedding Alignment for Enhanced Visual Representation in Vision-Language Models. (2025). arXiv:2506.02557"},{"key":"e_1_3_2_1_7_1","volume-title":"MMWorld: Towards Multi-discipline Multi-faceted World Model Evaluation in Videos. In The Thirteenth International Conference on Learning Representations (ICLR).","author":"He Xuehai","year":"2025","unstructured":"Xuehai He, Weixi Feng, Kaizhi Zheng, Yujie Lu, Wanrong Zhu, Jiachen Li, Yue Fan, Jianfeng Wang, Linjie Li, Zhengyuan Yang, et al., 2025. MMWorld: Towards Multi-discipline Multi-faceted World Model Evaluation in Videos. In The Thirteenth International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_8_1","unstructured":"Kairui Hu Penghao Wu Fanyi Pu Wang Xiao Yuanhan Zhang Xiang Yue Bo Li and Ziwei Liu. 2025. Video-MMMU: Evaluating Knowledge Acquisition from Multi-discipline Professional Videos. (2025). arXiv:2501.13826"},{"key":"e_1_3_2_1_9_1","volume-title":"Aria: An Open Multimodal Native Mixture-of-Experts Model.","author":"Li Dongxu","year":"2024","unstructured":"Dongxu Li, Yudong Liu, Haoning Wu, Yue Wang, Zhiqi Shen, Bowen Qu, Xinyao Niu, Guoyin Wang, Bei Chen, and Junnan Li. 2024. Aria: An Open Multimodal Native Mixture-of-Experts Model. (2024). arXiv:2410.05993"},{"key":"e_1_3_2_1_10_1","volume-title":"Shafiq Joty, Caiming Xiong, and Steven Hoi.","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Ramprasaath R. Selvaraju, Akhilesh Deepak Gotmare, Shafiq Joty, Caiming Xiong, and Steven Hoi. 2021. Align Before Fuse: Vision and Language Representation Learning with Momentum Distillation. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00315"},{"key":"e_1_3_2_1_13_1","unstructured":"Qwen Team. 2025. Qwen3 Technical Report. arXiv:2505.09388"},{"key":"e_1_3_2_1_14_1","volume-title":"International Conference on Machine Learning (ICML). 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning Transferable Visual Models from Natural Language Supervision. In International Conference on Machine Learning (ICML). 8748-8763."},{"key":"e_1_3_2_1_15_1","unstructured":"Hanoona Rasheed Abdelrahman Shaker Anqi Tang Muhammad Maaz Ming-Hsuan Yang Salman Khan and Fahad Shahbaz Khan. 2025. VideoMathQA: Benchmarking Mathematical Reasoning via Multimodal Understanding in Videos. (2025). arXiv:2506.05349"},{"key":"e_1_3_2_1_16_1","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Junxiao Song Xiao Bi Haowei Zhang Mingchuan Zhang Y. K. Li Yang Wu et al. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. (2024). arXiv:2402.03300"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2025.3527978"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02711"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"e_1_3_2_1_20_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar et al. 2017. Attention Is All You Need. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"e_1_3_2_1_23_1","unstructured":"Boqiang Zhang Kehan Li Zesen Cheng Zhiqiang Hu Yuqian Yuan Guanzheng Chen Sicong Leng Yuming Jiang Hang Zhang Xin Li Peng Jin Wenqi Zhang Fan Wang Lidong Bing and Deli Zhao. 2025. VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding. (2025). arXiv:2501.13106"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"e_1_3_2_1_25_1","unstructured":"Yuanhan Zhang Jinming Wu Wei Li Bo Li Zejun Ma Ziwei Liu and Chunyuan Li. 2024. Video Instruction Tuning with Synthetic Data. (2024). arXiv:2410.02713"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01024"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00793"},{"key":"e_1_3_2_1_28_1","volume-title":"Look Twice Before You Answer: Memory-Space Visual Retracing for Hallucination Mitigation in Multimodal Large Language Models. Forty-second International Conference on Machine Learning (ICML)","author":"Zou Xin","year":"2025","unstructured":"Xin Zou, Yizhou Wang, Yibo Yan, Yuanhuiyi Lyu, Kening Zheng, Sirui Huang, Junkai Chen, Peijie Jiang, Jia Liu, Chang Tang, and Xuming Hu. 2025. Look Twice Before You Answer: Memory-Space Visual Retracing for Hallucination Mitigation in Multimodal Large Language Models. Forty-second International Conference on Machine Learning (ICML) (2025)."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:26:32Z","timestamp":1784136392000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809586"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":28,"alternative-id":["10.1145\/3805712.3809586","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809586","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}