{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T19:35:11Z","timestamp":1781811311546,"version":"3.54.5"},"publisher-location":"Cham","reference-count":50,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031730061","type":"print"},{"value":"9783031730078","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73007-8_12","type":"book-chapter","created":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T19:02:40Z","timestamp":1727722960000},"page":"193-210","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Learning to\u00a0Localize Actions in\u00a0Instructional Videos with\u00a0LLM-Based Multi-pathway Text-Video Alignment"],"prefix":"10.1007","author":[{"given":"Yuxiao","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wentao","family":"Bao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deep","family":"Patel","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Kong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Martin Renqiang","family":"Min","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dimitris N.","family":"Metaxas","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,1]]},"reference":[{"key":"12_CR1","unstructured":"Ahn, M., et al.: Do as i can, not as i say: grounding language in robotic affordances. arXiv preprint arXiv:2204.01691 (2022)"},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1728\u20131738 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"12_CR3","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Caba\u00a0Heilbron, F., Escorcia, V., Ghanem, B., Carlos\u00a0Niebles, J.: ActivityNet: a large-scale video benchmark for human activity understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 961\u2013970 (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"12_CR5","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1007\/978-3-031-19809-0_3","volume-title":"European Conference on Computer Vision 2022","author":"M Cao","year":"2022","unstructured":"Cao, M., Yang, T., Weng, J., Zhang, C., Wang, J., Zou, Y.: LocVTP: video-text pre-training for temporal localization. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13686, pp. 38\u201356. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19809-0_3"},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Chang, C.Y., Huang, D.A., Sui, Y., Fei-Fei, L., Niebles, J.C.: D3TW: discriminative differentiable dynamic time warping for weakly supervised action alignment and segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3546\u20133555 (2019)","DOI":"10.1109\/CVPR.2019.00366"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Chen, B., et al.: Multimodal clustering networks for self-supervised learning from unlabeled videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8012\u20138021 (2021)","DOI":"10.1109\/ICCV48922.2021.00791"},{"key":"12_CR9","unstructured":"Chen, D., Liu, J., Dai, W., Wang, B.: Visual instruction tuning with polite flamingo. arXiv preprint arXiv:2307.01003 (2023)"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Chen, L., et al.: Weakly-supervised temporal article grounding. arXiv preprint arXiv:2210.12444 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.639"},{"key":"12_CR11","unstructured":"Dai, W., et al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. In: Oh, A., Naumann, T., Globerson, A., Saenko, K., Hardt, M., Levine, S. (eds.) Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, 10\u201316 December 2023 (2023). http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/9a6a435e75419a836fe47ab6793623e6-Abstract-Conference.html"},{"key":"12_CR12","unstructured":"Driess, D., et al.: PaLM-E: an embodied multimodal language model. arXiv preprint arXiv:2303.03378 (2023)"},{"key":"12_CR13","unstructured":"Dvornik, M., Hadji, I., Derpanis, K.G., Garg, A., Jepson, A.: Drop-DTW: aligning common signal between sequences while dropping outliers. In: Advances in Neural Information Processing Systems, vol. 34, pp. 13782\u201313793 (2021)"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Dvornik, N., Hadji, I., Zhang, R., Derpanis, K.G., Wildes, R.P., Jepson, A.D.: StepFormer: self-supervised step discovery and localization in instructional videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18952\u201318961 (2023)","DOI":"10.1109\/CVPR52729.2023.01817"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Elhamifar, E., Naing, Z.: Unsupervised procedure learning via joint dynamic summarization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6341\u20136350 (2019)","DOI":"10.1109\/ICCV.2019.00644"},{"issue":"3\u20134","key":"12_CR16","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1561\/0600000105","volume":"14","author":"Z Gan","year":"2022","unstructured":"Gan, Z., et al.: Vision-language pre-training: basics, recent advances, and future trends. Found. Trends\u00ae Comput. Graph. Vis. 14(3\u20134), 163\u2013352 (2022)","journal-title":"Found. Trends\u00ae Comput. Graph. Vis."},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Gilardi, F., Alizadeh, M., Kubli, M.: ChatGPT outperforms crowd-workers for text-annotation tasks. arXiv preprint arXiv:2303.15056 (2023)","DOI":"10.1073\/pnas.2305016120"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Han, T., Xie, W., Zisserman, A.: Temporal alignment networks for long-term video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2906\u20132916 (2022)","DOI":"10.1109\/CVPR52688.2022.00292"},{"key":"12_CR19","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"12_CR20","unstructured":"Kay, W., et al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Ko, D., et al.: Video-text representation learning via differentiable weak temporal alignment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5016\u20135025 (2022)","DOI":"10.1109\/CVPR52688.2022.00496"},{"key":"12_CR22","unstructured":"Koupaee, M., Wang, W.Y.: WikiHow: a large scale text summarization dataset. arXiv preprint arXiv:1810.09305 (2018)"},{"key":"12_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/978-3-319-49409-8_7","volume-title":"Computer Vision \u2013 ECCV 2016 Workshops","author":"C Lea","year":"2016","unstructured":"Lea, C., Vidal, R., Reiter, A., Hager, G.D.: Temporal convolutional networks: a unified approach to action segmentation. In: Hua, G., J\u00e9gou, H. (eds.) ECCV 2016, Part III. LNCS, vol. 9915, pp. 47\u201354. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-49409-8_7"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"Lin, W., et al.: Match, expand and improve: unsupervised finetuning for zero-shot action recognition with language knowledge. arXiv preprint arXiv:2303.08914 (2023)","DOI":"10.1109\/ICCV51070.2023.00267"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Lin, X., Petroni, F., Bertasius, G., Rohrbach, M., Chang, S.F., Torresani, L.: Learning to recognize procedural activities with distant supervision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13853\u201313863 (2022)","DOI":"10.1109\/CVPR52688.2022.01348"},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. arXiv preprint arXiv:2310.03744 (2023)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"12_CR27","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Lu, Z., Elhamifar, E.: Set-supervised action learning in procedural task videos via pairwise order consistency. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19903\u201319913 (2022)","DOI":"10.1109\/CVPR52688.2022.01928"},{"key":"12_CR29","unstructured":"Luo, H., et al.: UniVL: a unified video and language pre-training model for multimodal understanding and generation. arXiv preprint arXiv:2002.06353 (2020)"},{"key":"12_CR30","doi-asserted-by":"crossref","unstructured":"Mavroudi, E., Afouras, T., Torresani, L.: Learning to ground instructional articles in videos through narrations. arXiv preprint arXiv:2306.03802 (2023)","DOI":"10.1109\/ICCV51070.2023.01395"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J.B., Smaira, L., Laptev, I., Sivic, J., Zisserman, A.: End-to-end learning of visual representations from uncurated instructional videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9879\u20139889 (2020)","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"12_CR32","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.B., Tapaswi, M., Laptev, I., Sivic, J.: HowTo100M: learning a text-video embedding by watching hundred million narrated video clips. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2630\u20132640 (2019)","DOI":"10.1109\/ICCV.2019.00272"},{"key":"12_CR33","unstructured":"Pu, X., Gao, M., Wan, X.: Summarization is (almost) dead. arXiv preprint arXiv:2309.09558 (2023)"},{"key":"12_CR34","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"issue":"1","key":"12_CR35","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(1), 5485\u20135551 (2020)","journal-title":"J. Mach. Learn. Res."},{"issue":"1","key":"12_CR36","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/TASSP.1978.1163055","volume":"26","author":"H Sakoe","year":"1978","unstructured":"Sakoe, H., Chiba, S.: Dynamic programming algorithm optimization for spoken word recognition. IEEE Trans. Acoust. Speech Sig. Process. 26(1), 43\u201349 (1978)","journal-title":"IEEE Trans. Acoust. Speech Sig. Process."},{"key":"12_CR37","doi-asserted-by":"crossref","unstructured":"Shen, Y., Wang, L., Elhamifar, E.: Learning to segment actions from visual and language instructions via differentiable weak sequence alignment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10156\u201310165 (2021)","DOI":"10.1109\/CVPR46437.2021.01002"},{"key":"12_CR38","doi-asserted-by":"crossref","unstructured":"Shvetsova, N., Kukleva, A., Hong, X., Rupprecht, C., Schiele, B., Kuehne, H.: HowToCaption: prompting LLMS to transform video annotations at scale. arXiv preprint arXiv:2310.04900 (2023)","DOI":"10.1007\/978-3-031-72992-8_1"},{"key":"12_CR39","unstructured":"Tan, Z., et al.: Large language models for data annotation: a survey. arXiv preprint arXiv:2402.13446 (2024)"},{"key":"12_CR40","doi-asserted-by":"crossref","unstructured":"Tang, Y., et al.: COIN: a large-scale dataset for comprehensive instructional video analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1207\u20131216 (2019)","DOI":"10.1109\/CVPR.2019.00130"},{"key":"12_CR41","unstructured":"Touvron, H., et al.: LLaMA: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"12_CR42","unstructured":"Touvron, H., et al.: LLaMA 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"12_CR43","doi-asserted-by":"crossref","unstructured":"Van\u00a0Veen, D., et al.: Clinical text summarization: adapting large language models can outperform human experts. Res. Square (2023)","DOI":"10.21203\/rs.3.rs-3483777\/v1"},{"key":"12_CR44","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"12_CR45","unstructured":"Wang, Y., et al.: InternVideo: general video foundation models via generative and discriminative learning. arXiv preprint arXiv:2212.03191 (2022)"},{"key":"12_CR46","doi-asserted-by":"crossref","unstructured":"Xu, H., et al.: VideoCLIP: contrastive pre-training for zero-shot video-text understanding. arXiv preprint arXiv:2109.14084 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"12_CR47","unstructured":"Xue, H., et al.: CLIP-ViP: adapting pre-trained image-text model to video-language representation alignment. arXiv preprint arXiv:2209.06430 (2022)"},{"key":"12_CR48","doi-asserted-by":"crossref","unstructured":"Ye, J., et al.: ZeroGen: efficient zero-shot learning via dataset generation. arXiv preprint arXiv:2202.07922 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.801"},{"key":"12_CR49","unstructured":"Zhao, Q., et al.: AntGPT: can large language models help long-term action anticipation from videos? arXiv preprint arXiv:2307.16368 (2023)"},{"key":"12_CR50","doi-asserted-by":"crossref","unstructured":"Zhou, H., Mart\u00edn-Mart\u00edn, R., Kapadia, M., Savarese, S., Niebles, J.C.: Procedure-aware pretraining for instructional video understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10727\u201310738 (2023)","DOI":"10.1109\/CVPR52729.2023.01033"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73007-8_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T22:36:53Z","timestamp":1732833413000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73007-8_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,1]]},"ISBN":["9783031730061","9783031730078"],"references-count":50,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73007-8_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,1]]},"assertion":[{"value":"1 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}