{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T06:45:58Z","timestamp":1785653158063,"version":"3.56.0"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_44","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:43Z","timestamp":1785649603000},"page":"678-693","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["LLaVA-MR: Large Language-and-Vision Assistant for\u00a0Video Moment Retrieval"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-4376-5875","authenticated-orcid":false,"given":"Weiheng","family":"Lu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0242-6481","authenticated-orcid":false,"given":"Jian","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9835-4634","authenticated-orcid":false,"given":"An","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9325-5341","authenticated-orcid":false,"given":"Ming-Ching","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"44_CR1","unstructured":"Cao, Z., Zhang, B., Du, H., Yu, X., Li, X., Wang, S.: Flashvtg: feature layering and adaptive score handling network for video temporal grounding (2024). https:\/\/arxiv.org\/abs\/2412.13441"},{"key":"44_CR2","unstructured":"Chung, H.W., et al.: Scaling instruction-finetuned language models (2022). https:\/\/arxiv.org\/abs\/2210.11416"},{"key":"44_CR3","unstructured":"Feng, Y., Shi, Y., Liu, F., Yan, T.: Motion guided token compression for efficient masked video modeling (2024). arXiv:2402.18577 arXiv preprint"},{"key":"44_CR4","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: Tall: Temporal activity localization via language query. In: ICCV, pp. 5267\u20135275 (2017)","DOI":"10.1109\/ICCV.2017.563"},{"key":"44_CR5","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)"},{"key":"44_CR6","doi-asserted-by":"crossref","unstructured":"Jang, J., Park, J., Kim, J., Kwon, H., Sohn, K.: Knowing where to focus: event-aware transformer for video grounding (2023). https:\/\/arxiv.org\/abs\/2308.06947","DOI":"10.1109\/ICCV51070.2023.01273"},{"key":"44_CR7","unstructured":"Jin, Y., et\u00a0al.: Efficient multimodal large language models: a survey. arXiv preprint arXiv:2405.10739 (2024)"},{"key":"44_CR8","doi-asserted-by":"crossref","unstructured":"Kim, M., Kim, H.B., Moon, J., Choi, J., Kim, S.T.: Do you remember? dense video captioning with cross-modal memory retrieval (2024). https:\/\/arxiv.org\/abs\/2404.07610","DOI":"10.1109\/CVPR52733.2024.01318"},{"key":"44_CR9","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., Niebles, J.C.: Dense-captioning events in videos (2017). https:\/\/arxiv.org\/abs\/1705.00754","DOI":"10.1109\/ICCV.2017.83"},{"key":"44_CR10","first-page":"11846","volume":"34","author":"J Lei","year":"2021","unstructured":"Lei, J., Berg, T.L., Bansal, M.: Detecting moments and highlights in videos via natural language queries. NeurIPS 34, 11846\u201311858 (2021)","journal-title":"NeurIPS"},{"key":"44_CR11","unstructured":"Li, J., et al.: A survey on benchmarks of multimodal large language models (2024). https:\/\/arxiv.org\/abs\/2408.08632"},{"key":"44_CR12","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models (2023). https:\/\/arxiv.org\/abs\/2301.12597"},{"key":"44_CR13","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"44_CR14","doi-asserted-by":"crossref","unstructured":"Lin, K.Q., et al.: Univtg: towards unified video-language temporal grounding (2023). https:\/\/arxiv.org\/abs\/2307.16715","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"44_CR15","unstructured":"Liu, W., Miao, B., Cao, J., Zhu, X., Liu, B., Nasim, M., Mian, A.: Context-enhanced video moment retrieval with large language models (2024). https:\/\/arxiv.org\/abs\/2405.12540"},{"key":"44_CR16","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: $$r^2$$-tuning: efficient image-to-video transfer learning for video temporal grounding (2024). https:\/\/arxiv.org\/abs\/2404.00801","DOI":"10.1007\/978-3-031-72940-9_24"},{"key":"44_CR17","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization (2019). https:\/\/arxiv.org\/abs\/1711.05101"},{"key":"44_CR18","unstructured":"Meinardus, B., Batra, A., Rohrbach, A., Rohrbach, M.: The surprising effectiveness of multimodal large language models for video moment retrieval (2024). https:\/\/arxiv.org\/abs\/2406.18113"},{"key":"44_CR19","unstructured":"Moon, W., Hyun, S., Lee, S., Heo, J.P.: Correlation-guided query-dependency calibration for video temporal grounding (2024). https:\/\/arxiv.org\/abs\/2311.08835"},{"key":"44_CR20","doi-asserted-by":"crossref","unstructured":"Moon, W., Hyun, S., Park, S., Park, D., Heo, J.P.: Query-dependent video representation for moment retrieval and highlight detection (2023). https:\/\/arxiv.org\/abs\/2303.13874","DOI":"10.1109\/CVPR52729.2023.02205"},{"key":"44_CR21","unstructured":"Paul, D., Parvez, M.R., Mohammed, N., Rahman, S.: Videolights: feature refinement and cross-task alignment transformer for joint video highlight detection and moment retrieval (2024). https:\/\/arxiv.org\/abs\/2412.01558"},{"key":"44_CR22","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"44_CR23","doi-asserted-by":"crossref","unstructured":"Soldan, M., Xu, M., Qu, S., Tegner, J., Ghanem, B.: Vlg-net: Video-language graph matching network for video grounding (2021). https:\/\/arxiv.org\/abs\/2011.10132","DOI":"10.1109\/ICCVW54120.2021.00361"},{"key":"44_CR24","unstructured":"Sun, G., et al.: video-salmonn: Speech-enhanced audio-visual large language models (2024). https:\/\/arxiv.org\/abs\/2406.15704"},{"key":"44_CR25","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Dynamic-vlm: simple dynamic visual token compression for videollm. arXiv preprint arXiv:2412.09530 (2024)","DOI":"10.1109\/ICCV51701.2025.01935"},{"key":"44_CR26","unstructured":"Wang, T., Zhang, J., Zheng, F., Jiang, W., Cheng, R., Luo, P.: Learning grounded vision-language representation for versatile understanding in untrimmed videos (2023). https:\/\/arxiv.org\/abs\/2303.06378"},{"key":"44_CR27","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, Y., Zohar, O., Yeung-Levy, S.: Videoagent: Long-form video understanding with large language model as agent (2024). https:\/\/arxiv.org\/abs\/2403.10517","DOI":"10.1007\/978-3-031-72989-8_4"},{"key":"44_CR28","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Internvideo2: scaling foundation models for multimodal video understanding (2024). https:\/\/arxiv.org\/abs\/2403.15377","DOI":"10.1007\/978-3-031-73013-9_23"},{"key":"44_CR29","doi-asserted-by":"crossref","unstructured":"Wu, Y., et al.: Number it: Temporal grounding videos like flipping manga (2025). https:\/\/arxiv.org\/abs\/2411.10332","DOI":"10.1109\/CVPR52734.2025.01284"},{"key":"44_CR30","unstructured":"Xiao, Y., et al.: Bridging the gap: a unified video comprehension framework for moment retrieval and highlight detection (2023). https:\/\/arxiv.org\/abs\/2311.16464"},{"key":"44_CR31","unstructured":"Yan, S., et al.: Unloc: a unified framework for video localization tasks (2023). https:\/\/arxiv.org\/abs\/2308.11062"},{"key":"44_CR32","doi-asserted-by":"crossref","unstructured":"Yu, S., Cho, J., Yadav, P., Bansal, M.: Self-chained image-language model for video localization and question answering (2023). https:\/\/arxiv.org\/abs\/2305.06988","DOI":"10.52202\/075280-3354"},{"key":"44_CR33","doi-asserted-by":"crossref","unstructured":"Zeng, R., Xu, H., Huang, W., Chen, P., Tan, M., Gan, C.: Dense regression network for video grounding (2020). https:\/\/arxiv.org\/abs\/2004.03545","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"44_CR34","doi-asserted-by":"crossref","unstructured":"Zeng, Y., Zhong, Y., Feng, C., Ma, L.: Unimd: Towards unifying moment retrieval and temporal action detection (2024). https:\/\/arxiv.org\/abs\/2404.04933","DOI":"10.1007\/978-3-031-72952-2_17"},{"key":"44_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, H., Li, X., Bing, L.: Video-llama: An instruction-tuned audio-visual language model for video understanding (2023). https:\/\/arxiv.org\/abs\/2306.02858","DOI":"10.18653\/v1\/2023.emnlp-demo.49"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_44","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:46:47Z","timestamp":1785649607000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_44"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_44","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}