{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T08:02:33Z","timestamp":1784361753286,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234288","type":"print"},{"value":"9789819234295","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3429-5_9","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:06:39Z","timestamp":1784358399000},"page":"103-114","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["LeAdQA: LLM-Driven Context-Aware Temporal Grounding for Video Question Answering"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-3083-3323","authenticated-orcid":false,"given":"Xinxin","family":"Dong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5240-1779","authenticated-orcid":false,"given":"Baoyun","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haokai","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yufei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zixuan","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaodong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"9_CR1","doi-asserted-by":"crossref","unstructured":"Wang, Z., et al.: VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos. arXiv:2405.19209 (2024)","DOI":"10.1109\/CVPR52734.2025.00311"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Xiao, J., Shang, X., Yao, A., Chua, T.-S.: NExT-QA: Next phase of question-answering to explaining temporal actions. arXiv:2105.08276 (2021)","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Bansal, M., Berg, T.: TVQA: Localized, compositional video question answering. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 1369\u20131379. Association for Computational Linguistics, Brussels (2018)","DOI":"10.18653\/v1\/D18-1167"},{"key":"9_CR4","doi-asserted-by":"crossref","unstructured":"Zhang, C., et al.: A simple LLM framework for long-range video question-answering. In: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 21715\u201321737. Association for Computational Linguistics, Miami (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.1209"},{"key":"9_CR5","unstructured":"Guo, Y., Liu, J., Li, M., Liu, Q., Chen, X., Tang, X.: TRACE: Temporal grounding video LLM via causal event modeling. arXiv:2410.05643 (2024)"},{"key":"9_CR6","unstructured":"Wang, X., Si, Q., Wu, J., Zhu, S., Cao, L., Nie, L.: ReTaKe: Reducing temporal and knowledge redundancy for long video understanding. arXiv:2412.20504 (2025)"},{"key":"9_CR7","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: TALL: Temporal activity localization via language query. arXiv:1705.02101 (2017)","DOI":"10.1109\/ICCV.2017.563"},{"key":"9_CR8","unstructured":"Lei, J., Berg, T.L., Bansal, M.: QVHighlights: Detecting moments and highlights in videos via natural language queries. arXiv:2107.09609 (2021)"},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Moon, W., Hyun, S., Park, S., Park, D., Heo, J.-P.: Query-dependent video representation for moment retrieval and highlight detection. arXiv:2303.13874 (2023)","DOI":"10.1109\/CVPR52729.2023.02205"},{"key":"9_CR10","doi-asserted-by":"publisher","first-page":"386","DOI":"10.1016\/j.neucom.2018.06.069","volume":"314","author":"W Chu","year":"2018","unstructured":"Chu, W., Xue, H., Zhao, Z., Cai, D., Yao, C.: The forgettable-watcher model for video question answering. Neurocomputing. 314, 386\u2013393 (2018)","journal-title":"Neurocomputing"},{"key":"9_CR11","doi-asserted-by":"publisher","first-page":"931","DOI":"10.1109\/TCSVT.2020.2995959","volume":"31","author":"T Yu","year":"2021","unstructured":"Yu, T., Yu, J., Yu, Z., Huang, Q., Tian, Q.: Long-term video question answering via multimodal hierarchical memory attentive networks. IEEE Trans. Circuits Syst. Video Technol. 31, 931\u2013944 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"9_CR12","doi-asserted-by":"crossref","unstructured":"Tan, R., et al.: Koala: Key frame-conditioned long video-LLM. arXiv:2404.04346 (2024)","DOI":"10.1109\/CVPR52733.2024.01289"},{"key":"9_CR13","unstructured":"Fei, H., et al.: Video-of-thought: step-by-step video reasoning from perception to cognition. In: Proceedings of the 41st International Conference on Machine Learning (ICML 2024) (2024)"},{"key":"9_CR14","doi-asserted-by":"publisher","unstructured":"Wang, J., Yuan, L., Zhang, Y., Sun, H.: Tarsier: Recipes for training and evaluating large video description models. arXiv. https:\/\/doi.org\/10.48550\/arXiv.2407.00634 (2024)","DOI":"10.48550\/arXiv.2407.00634"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, Y., Zohar, O., Yeung-Levy, S.: VideoAgent: Long-form video understanding with large language model as agent. arXiv:2403.10517 (2024)","DOI":"10.1007\/978-3-031-72989-8_4"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"Ahmad, W., Peng, Y.-T., Chang, Y.-H., Ganfure, G.O., Khan, S.: CapST: Leveraging capsule networks and temporal attention for accurate model attribution in deep-fake videos. http:\/\/arxiv.org\/abs\/2311.03782 (2025)","DOI":"10.1145\/3715138"},{"key":"9_CR17","doi-asserted-by":"crossref","unstructured":"Lin, K.Q., et al.: UniVTG: Towards unified video-language temporal grounding. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2782\u20132792. IEEE, Paris (2023)","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"9_CR18","unstructured":"Bai, S., et al.: Qwen2.5-VL technical report. https:\/\/arxiv.org\/abs\/2502.13923 (2025)"},{"key":"9_CR19","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. http:\/\/arxiv.org\/abs\/2103.00020 (2021)"},{"key":"9_CR20","doi-asserted-by":"crossref","unstructured":"Li, J., Wei, P., Han, W., Fan, L.: IntentQA: Context-aware video intent reasoning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV 2023), pp. 11963\u201311974 (2023)","DOI":"10.1109\/ICCV51070.2023.01099"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Xiao, J., Yao, A., Li, Y., Chua, T.-S.: Can I trust your answer? Visually grounded video question answering. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13204\u201313214. IEEE, Seattle, WA, USA (2024)","DOI":"10.1109\/CVPR52733.2024.01254"},{"key":"9_CR22","unstructured":"OpenAI, Achiam, J., et al.: GPT-4 Technical Report. arXiv:2303.08774 (2024)"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Li, Y., Wang, X., Xiao, J., Ji, W., Chua, T.-S.: Invariant grounding for video question answering. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2918\u20132927. IEEE, New Orleans, LA, USA (2022)","DOI":"10.1109\/CVPR52688.2022.00294"},{"key":"9_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2024.127905","volume":"596","author":"X Jing","year":"2024","unstructured":"Jing, X., Yang, G., Chu, J.: An empirical study of excitation and aggregation design adaptions in CLIP4Clip for video-text retrieval. Neurocomputing. 596, 127905 (2024)","journal-title":"Neurocomputing"},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Choudhury, R., Niinuma, K., Kitani, K.M., Jeni, L.A.: Zero-shot video question answering with procedural programs. arXiv:2312.00937 (2023)","DOI":"10.1007\/978-3-031-72920-1_18"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Yu, S., Cho, J., Yadav, P., Bansal, M.: Self-chained image-language model for video localization and question answering. In: Advances in Neural Information Processing Systems 36 (NeurIPS 2023) (2023)","DOI":"10.52202\/075280-3354"},{"key":"9_CR27","unstructured":"Wang, Y., et al.: InternVideo: General video foundation models via generative and discriminative learning. arXiv:2212.03191 (2022)"},{"key":"9_CR28","unstructured":"Ranasinghe, K., Li, X., Kahatapitiya, K., Ryoo, M.S.: Understanding long videos with multimodal language models. arXiv:2403.16998 (2025)"},{"key":"9_CR29","doi-asserted-by":"crossref","unstructured":"Kim, W., Choi, C., Lee, W., Rhee, W.: An image grid can be worth a video: zero-shot video question answering using a VLM. arXiv:2403.18406 (2024)","DOI":"10.1109\/ACCESS.2024.3517625"},{"key":"9_CR30","doi-asserted-by":"crossref","unstructured":"Kahatapitiya, K., Ranasinghe, K., Park, J., Ryoo, M.S.: Language repository for long video understanding. In: Findings of the Association for Computational Linguistics: ACL 2025, pp. 5627\u20135646. Association for Computational Linguistics (2025)","DOI":"10.18653\/v1\/2025.findings-acl.294"},{"key":"9_CR31","unstructured":"Park, J., Ranasinghe, K., Kahatapitiya, K., Ryu, W., Kim, D., Ryoo, M.S.: Too many frames, not all useful: Efficient strategies for long-form videoQA. arXiv:2406.09396 (2024)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3429-5_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:06:41Z","timestamp":1784358401000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3429-5_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819234288","9789819234295"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3429-5_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}