{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T17:18:03Z","timestamp":1770830283740,"version":"3.50.1"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T00:00:00Z","timestamp":1736121600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T00:00:00Z","timestamp":1736121600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272083"],"award-info":[{"award-number":["62272083"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62306059"],"award-info":[{"award-number":["62306059"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62102061"],"award-info":[{"award-number":["62102061"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Liaoning Provincial NSF","award":["2022-MS-128"],"award-info":[{"award-number":["2022-MS-128"]}]},{"name":"Liaoning Provincial NSF","award":["2022-MS-137"],"award-info":[{"award-number":["2022-MS-137"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s00530-024-01639-8","type":"journal-article","created":{"date-parts":[[2025,1,6]],"date-time":"2025-01-06T12:01:18Z","timestamp":1736164878000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Temporal refinement and multi-grained matching for moment retrieval and highlight detection"],"prefix":"10.1007","volume":"31","author":[{"given":"Cunjuan","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanyi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Jia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weimin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,6]]},"reference":[{"key":"1639_CR1","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: Tall: Temporal activity localization via language query. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5267\u20135275 (2017)","DOI":"10.1109\/ICCV.2017.563"},{"key":"1639_CR2","doi-asserted-by":"crossref","unstructured":"Ji, W., Liang, R., Zheng, Z., Zhang, W., Zhang, S., Li, J., Li, M., Chua, T.-s.: Are binary annotations sufficient? video moment retrieval via hierarchical uncertainty-based active learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23013\u201323022 (2023)","DOI":"10.1109\/CVPR52729.2023.02204"},{"key":"1639_CR3","doi-asserted-by":"crossref","unstructured":"Luo, D., Huang, J., Gong, S., Jin, H., Liu, Y.: Towards generalisable video moment retrieval: Visual-dynamic injection to image-text pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23045\u201323055 (2023)","DOI":"10.1109\/CVPR52729.2023.02207"},{"key":"1639_CR4","doi-asserted-by":"crossref","unstructured":"Sun, X., Gao, J., Zhu, Y., Wang, X., Zhou, X.: Video moment retrieval via comprehensive relation-aware network. IEEE Trans. Circ. Syst. Video Technol. (2023)","DOI":"10.1109\/TCSVT.2023.3250518"},{"key":"1639_CR5","doi-asserted-by":"crossref","unstructured":"Gao, J., Xu, C.: Fast video moment retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1523\u20131532 (2021)","DOI":"10.1109\/ICCV48922.2021.00155"},{"issue":"3","key":"1639_CR6","doi-asserted-by":"publisher","first-page":"1646","DOI":"10.1109\/TCSVT.2021.3075470","volume":"32","author":"J Gao","year":"2021","unstructured":"Gao, J., Xu, C.: Learning video moment retrieval without a single annotated video. IEEE Trans. Circ. Syst. Video Technol. 32(3), 1646\u20131657 (2021)","journal-title":"IEEE Trans. Circ. Syst. Video Technol."},{"key":"1639_CR7","doi-asserted-by":"crossref","unstructured":"Sun, M., Farhadi, A., Seitz, S.: Ranking domain-specific highlights by analyzing edited videos. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part I 13, pp. 787\u2013802 (2014). Springer","DOI":"10.1007\/978-3-319-10590-1_51"},{"key":"1639_CR8","doi-asserted-by":"crossref","unstructured":"Yao, T., Mei, T., Rui, Y.: Highlight detection with pairwise deep ranking for first-person video summarization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 982\u2013990 (2016)","DOI":"10.1109\/CVPR.2016.112"},{"key":"1639_CR9","doi-asserted-by":"crossref","unstructured":"Wei, F., Wang, B., Ge, T., Jiang, Y., Li, W., Duan, L.: Learning pixel-level distinctions for video highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3073\u20133082 (2022)","DOI":"10.1109\/CVPR52688.2022.00308"},{"key":"1639_CR10","first-page":"11846","volume":"34","author":"J Lei","year":"2021","unstructured":"Lei, J., Berg, T.L., Bansal, M.: Detecting moments and highlights in videos via natural language queries. Adv. Neural. Inf. Process. Syst. 34, 11846\u201311858 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1639_CR11","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229 (2020). Springer","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1639_CR12","doi-asserted-by":"crossref","unstructured":"Moon, W., Hyun, S., Park, S., Park, D., Heo, J.-P.: Query-dependent video representation for moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23023\u201323033 (2023)","DOI":"10.1109\/CVPR52729.2023.02205"},{"key":"1639_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, S., Wu, Y., Chen, C.-W., Shan, Y., Qie, X.: Umt: Unified multi-modal transformers for joint video moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3042\u20133051 (2022)","DOI":"10.1109\/CVPR52688.2022.00305"},{"key":"1639_CR14","doi-asserted-by":"crossref","unstructured":"Lin, K.Q., Zhang, P., Chen, J., Pramanick, S., Gao, D., Wang, A.J., Yan, R., Shou, M.Z.: Univtg: Towards unified video-language temporal grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2794\u20132804 (2023)","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"1639_CR15","doi-asserted-by":"crossref","unstructured":"Song, Y., Vallmitjana, J., Stent, A., Jaimes, A.: Tvsum: Summarizing web videos using titles. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5179\u20135187 (2015)","DOI":"10.1109\/CVPR.2015.7299154"},{"key":"1639_CR16","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: A joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1728\u20131738 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"1639_CR17","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","volume":"508","author":"H Luo","year":"2022","unstructured":"Luo, H., Ji, L., Zhong, M., Chen, Y., Lei, W., Duan, N., Li, T.: Clip4clip: An empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing 508, 293\u2013304 (2022)","journal-title":"Neurocomputing"},{"issue":"1","key":"1639_CR18","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s13735-023-00267-8","volume":"12","author":"C Zhu","year":"2023","unstructured":"Zhu, C., Jia, Q., Chen, W., Guo, Y., Liu, Y.: Deep learning for video-text retrieval: a review. Int. J. Multimed. Inf. Retrieval 12(1), 3 (2023)","journal-title":"Int. J. Multimed. Inf. Retrieval"},{"key":"1639_CR19","doi-asserted-by":"crossref","unstructured":"Jang, J., Park, J., Kim, J., Kwon, H., Sohn, K.: Knowing where to focus: Event-aware transformer for video grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13846\u201313856 (2023)","DOI":"10.1109\/ICCV51070.2023.01273"},{"key":"1639_CR20","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Ma, L., Zhu, W.: Sentence specified dynamic video thumbnail generation. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 2332\u20132340 (2019)","DOI":"10.1145\/3343031.3350985"},{"key":"1639_CR21","doi-asserted-by":"crossref","unstructured":"Xiao, S., Chen, L., Zhang, S., Ji, W., Shao, J., Ye, L., Xiao, J.: Boundary proposal network for two-stage natural language video localization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 2986\u20132994 (2021)","DOI":"10.1609\/aaai.v35i4.16406"},{"key":"1639_CR22","doi-asserted-by":"crossref","unstructured":"Xu, H., He, K., Plummer, B.A., Sigal, L., Sclaroff, S., Saenko, K.: Multilevel language and vision integration for text-to-clip retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 9062\u20139069 (2019)","DOI":"10.1609\/aaai.v33i01.33019062"},{"key":"1639_CR23","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, X., Xu, M., Zhou, X., Ghanem, B.: Relation-aware video reading comprehension for temporal language grounding. arXiv preprint arXiv:2110.05717 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.324"},{"key":"1639_CR24","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Mei, T., Zhu, W.: To find where you talk: Temporal sentence localization in video with attention based location regression. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 9159\u20139166 (2019)","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"1639_CR25","doi-asserted-by":"crossref","unstructured":"Mun, J., Cho, M., Han, B.: Local-global video-text interactions for temporal grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10810\u201310819 (2020)","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"1639_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, H., Sun, A., Jing, W., Zhou, J.T.: Span-based localizing network for natural language video localization. arXiv preprint arXiv:2004.13931 (2020)","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"1639_CR27","unstructured":"Miech, A., Laptev, I., Sivic, J.: Learning a text-video embedding from incomplete and heterogeneous data. arXiv preprint arXiv:1804.02516 (2018)"},{"key":"1639_CR28","doi-asserted-by":"crossref","unstructured":"Gabeur, V., Sun, C., Alahari, K., Schmid, C.: Multi-modal transformer for video retrieval. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IV 16, pp. 214\u2013229 (2020). Springer","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"1639_CR29","doi-asserted-by":"crossref","unstructured":"Ge, Y., Ge, Y., Liu, X., Wang, J., Wu, J., Shan, Y., Qie, X., Luo, P.: Miles: Visual bert pre-training with injected language semantics for video-text retrieval. In: European Conference on Computer Vision, pp. 691\u2013708 (2022). Springer","DOI":"10.1007\/978-3-031-19833-5_40"},{"key":"1639_CR30","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhu, L., Yang, Y.: T2vlad: global-local sequence alignment for text-video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5079\u20135088 (2021)","DOI":"10.1109\/CVPR46437.2021.00504"},{"key":"1639_CR31","doi-asserted-by":"crossref","unstructured":"Dong, J., Li, X., Xu, C., Ji, S., He, Y., Yang, G., Wang, X.: Dual encoding for zero-example video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9346\u20139355 (2019)","DOI":"10.1109\/CVPR.2019.00957"},{"key":"1639_CR32","unstructured":"Fang, H., Xiong, P., Xu, L., Chen, Y.: Clip2video: Mastering video-text retrieval via image clip. arXiv preprint arXiv:2106.11097 (2021)"},{"key":"1639_CR33","doi-asserted-by":"crossref","unstructured":"Zheng, H., Pu, N., Li, W., Sebe, N., Zhong, Z.: Textual knowledge matters: Cross-modality co-teaching for generalized visual class discovery. arXiv preprint arXiv:2403.07369 (2024)","DOI":"10.1007\/978-3-031-72943-0_3"},{"key":"1639_CR34","doi-asserted-by":"crossref","unstructured":"Lao, M., Pu, N., Liu, Y., He, K., Bakker, E.M., Lew, M.S.: Coca: Collaborative causal regularization for audio-visual question answering. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 12995\u201313003 (2023)","DOI":"10.1609\/aaai.v37i11.26527"},{"key":"1639_CR35","doi-asserted-by":"crossref","unstructured":"Wu, P., He, X., Tang, M., Lv, Y., Liu, J.: Hanet: Hierarchical alignment networks for video-text retrieval. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 3518\u20133527 (2021)","DOI":"10.1145\/3474085.3475515"},{"key":"1639_CR36","doi-asserted-by":"crossref","unstructured":"Ma, Y., Xu, G., Sun, X., Yan, M., Zhang, J., Ji, R.: X-clip: End-to-end multi-grained contrastive learning for video-text retrieval. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 638\u2013647 (2022)","DOI":"10.1145\/3503161.3547910"},{"key":"1639_CR37","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Jia, Q., Fan, X., Liu, Y., He, R.: Cscnet: Class-specified cascaded network for compositional zero-shot learning. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 3705\u20133709 (2024)","DOI":"10.1109\/ICASSP48485.2024.10446756"},{"key":"1639_CR38","doi-asserted-by":"crossref","unstructured":"Liu, Y., Cai, Y., Jia, Q., Qiu, B., Wang, W., Pu, N.: Novel class discovery for ultra-fine-grained visual categorization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 17679\u201317688 (2024)","DOI":"10.1109\/CVPR52733.2024.01674"},{"key":"1639_CR39","doi-asserted-by":"crossref","unstructured":"Lao, M., Pu, N., Zhong, Z., Sebe, N., Lew, M.S.: Fedvqa: Personalized federated visual question answering over heterogeneous scenes. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 7796\u20137807 (2023)","DOI":"10.1145\/3581783.3611958"},{"key":"1639_CR40","unstructured":"Wang, X., Chen, H., Tang, S., Wu, Z., Zhu, W.: Disentangled representation learning. arXiv preprint arXiv:2211.11695 (2022)"},{"key":"1639_CR41","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. In: Advances in Neural Information Processing Systems (NeurIPS) (2017)"},{"key":"1639_CR42","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. International Conference on Learning Representations (ICLR) (2021)"},{"key":"1639_CR43","doi-asserted-by":"crossref","unstructured":"Xiong, B., Kalantidis, Y., Ghadiyaram, D., Grauman, K.: Less is more: Learning highlight detection from video duration. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1258\u20131267 (2019)","DOI":"10.1109\/CVPR.2019.00135"},{"key":"1639_CR44","doi-asserted-by":"crossref","unstructured":"Wang, L., Liu, D., Puri, R., Metaxas, D.N.: Learning trailer moments in full-length movies with co-contrastive attention. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVIII 16, pp. 300\u2013316 (2020). Springer","DOI":"10.1007\/978-3-030-58523-5_18"},{"key":"1639_CR45","doi-asserted-by":"crossref","unstructured":"Xu, M., Wang, H., Ni, B., Zhu, R., Sun, Z., Wang, C.: Cross-category video highlight detection via set-based learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7970\u20137979 (2021)","DOI":"10.1109\/ICCV48922.2021.00787"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01639-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01639-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01639-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,28]],"date-time":"2025-02-28T11:07:34Z","timestamp":1740740854000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01639-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,6]]},"references-count":45,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["1639"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01639-8","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,6]]},"assertion":[{"value":"14 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"46"}}