{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T19:43:54Z","timestamp":1743104634022,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":34,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819787913"},{"type":"electronic","value":"9789819787920"}],"license":[{"start":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T00:00:00Z","timestamp":1731110400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T00:00:00Z","timestamp":1731110400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8792-0_29","type":"book-chapter","created":{"date-parts":[[2024,11,8]],"date-time":"2024-11-08T06:56:13Z","timestamp":1731048973000},"page":"416-429","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MS-DETR: Exploiting Modality Synergy for\u00a0Moment Retrieval and\u00a0Highlight Detection"],"prefix":"10.1007","author":[{"given":"Luyuan","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Kong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tian","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianwu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,9]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Anne\u00a0Hendricks, L., Wang, O., Shechtman, E., Sivic, J., Darrell, T., Russell, B.: Localizing moments in video with natural language. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5803\u20135812 (2017)","key":"29_CR1","DOI":"10.1109\/ICCV.2017.618"},{"doi-asserted-by":"crossref","unstructured":"Badamdorj, T., Rochan, M., Wang, Y., Cheng, L.: Joint visual and audio learning for video highlight detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8127\u20138137 (2021)","key":"29_CR2","DOI":"10.1109\/ICCV48922.2021.00802"},{"doi-asserted-by":"crossref","unstructured":"Badamdorj, T., Rochan, M., Wang, Y., Cheng, L.: Contrastive learning for unsupervised video highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14042\u201314052 (2022)","key":"29_CR3","DOI":"10.1109\/CVPR52688.2022.01365"},{"doi-asserted-by":"crossref","unstructured":"Cai, S., Zuo, W., Davis, L.S., Zhang, L.: Weakly-supervised video summarization using variational encoder-decoder and web prior. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 184\u2013200 (2018)","key":"29_CR4","DOI":"10.1007\/978-3-030-01264-9_12"},{"doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer, Berlin (2020)","key":"29_CR5","DOI":"10.1007\/978-3-030-58452-8_13"},{"doi-asserted-by":"crossref","unstructured":"Chen, J., Chen, X., Ma, L., Jie, Z., Chua, T.S.: Temporally grounding natural sentence in video. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 162\u2013171 (2018)","key":"29_CR6","DOI":"10.18653\/v1\/D18-1015"},{"key":"29_CR7","first-page":"28442","volume":"34","author":"YW Chen","year":"2021","unstructured":"Chen, Y.W., Tsai, Y.H., Yang, M.H.: End-to-end multi-modal video temporal grounding. Adv. Neural. Inf. Process. Syst. 34, 28442\u201328453 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: Tall: Temporal activity localization via language query. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5267\u20135275 (2017)","key":"29_CR8","DOI":"10.1109\/ICCV.2017.563"},{"doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: Tall: temporal activity localization via language query. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5267\u20135275 (2017)","key":"29_CR9","DOI":"10.1109\/ICCV.2017.563"},{"unstructured":"Ghosh, S., Agarwal, A., Parekh, Z., Hauptmann, A.: Excl: extractive clip localization using natural language descriptions (2019). arXiv:1904.02755","key":"29_CR10"},{"unstructured":"Grauman, K., Westbury, A., Byrne, E., Chavis, Z., Furnari, A., Girdhar, R., Hamburger, J., Jiang, H., Liu, M., Liu, X., et\u00a0al.: Ego4d: around the world in 3,000 hours of egocentric video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18995\u201319012 (2022)","key":"29_CR11"},{"doi-asserted-by":"crossref","unstructured":"Gygli, M., Song, Y., Cao, L.: Video2gif: Automatic generation of animated gifs from video. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1001\u20131009 (2016)","key":"29_CR12","DOI":"10.1109\/CVPR.2016.114"},{"doi-asserted-by":"crossref","unstructured":"Huang, K.C., Wu, T.H., Su, H.T., Hsu, W.H.: MonoDTR: monocular 3D object detection with depth-aware transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4012\u20134021 (2022)","key":"29_CR13","DOI":"10.1109\/CVPR52688.2022.00398"},{"doi-asserted-by":"crossref","unstructured":"Khosla, A., Hamid, R., Lin, C.J., Sundaresan, N.: Large-scale video summarization using web-image priors. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2698\u20132705 (2013)","key":"29_CR14","DOI":"10.1109\/CVPR.2013.348"},{"key":"29_CR15","first-page":"11846","volume":"34","author":"J Lei","year":"2021","unstructured":"Lei, J., Berg, T.L., Bansal, M.: Detecting moments and highlights in videos via natural language queries. Adv. Neural. Inf. Process. Syst. 34, 11846\u201311858 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, S., Wu, Y., Chen, C.W., Shan, Y., Qie, X.: UMT: unified multi-modal transformers for joint video moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3042\u20133051 (2022)","key":"29_CR16","DOI":"10.1109\/CVPR52688.2022.00305"},{"doi-asserted-by":"crossref","unstructured":"Mahasseni, B., Lam, M., Todorovic, S.: Unsupervised video summarization with adversarial lstm networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 202\u2013211 (2017)","key":"29_CR17","DOI":"10.1109\/CVPR.2017.318"},{"doi-asserted-by":"crossref","unstructured":"Moon, W., Hyun, S., Park, S., Park, D., Heo, J.P.: Query-dependent video representation for moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23023\u201323033 (2023)","key":"29_CR18","DOI":"10.1109\/CVPR52729.2023.02205"},{"doi-asserted-by":"crossref","unstructured":"Panda, R., Das, A., Wu, Z., Ernst, J., Roy-Chowdhury, A.K.: Weakly supervised summarization of web videos. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3657\u20133666 (2017)","key":"29_CR19","DOI":"10.1109\/ICCV.2017.395"},{"unstructured":"Qinghong\u00a0Lin, K., Zhang, P., Chen, J., Pramanick, S., Gao, D., Jinpeng\u00a0Wang, A., Yan, R., Shou, M.Z.: UniVTG: towards unified video-language temporal grounding 2307 (2023)","key":"29_CR20"},{"key":"29_CR21","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1162\/tacl_a_00207","volume":"1","author":"M Regneri","year":"2013","unstructured":"Regneri, M., Rohrbach, M., Wetzel, D., Thater, S., Schiele, B., Pinkal, M.: Grounding action descriptions in videos. Trans. Assoc. Comput. Linguist. 1, 25\u201336 (2013)","journal-title":"Trans. Assoc. Comput. Linguist."},{"doi-asserted-by":"crossref","unstructured":"Song, Y., Vallmitjana, J., Stent, A., Jaimes, A.: Tvsum: Summarizing web videos using titles. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5179\u20135187 (2015)","key":"29_CR22","DOI":"10.1109\/CVPR.2015.7299154"},{"doi-asserted-by":"crossref","unstructured":"Sun, M., Farhadi, A., Seitz, S.: Ranking domain-specific highlights by analyzing edited videos. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part I 13, pp. 787\u2013802. Springer, Berlin (2014)","key":"29_CR23","DOI":"10.1007\/978-3-319-10590-1_51"},{"doi-asserted-by":"crossref","unstructured":"Sun, M., Farhadi, A., Seitz, S.: Ranking domain-specific highlights by analyzing edited videos. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part I 13, pp. 787\u2013802. Springer, Berlin (2014)","key":"29_CR24","DOI":"10.1007\/978-3-319-10590-1_51"},{"doi-asserted-by":"crossref","unstructured":"Xu, M., Wang, H., Ni, B., Zhu, R., Sun, Z., Wang, C.: Cross-category video highlight detection via set-based learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7970\u20137979 (2021)","key":"29_CR25","DOI":"10.1109\/ICCV48922.2021.00787"},{"doi-asserted-by":"crossref","unstructured":"Xu, M., Wang, H., Ni, B., Zhu, R., Sun, Z., Wang, C.: Cross-category video highlight detection via set-based learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7970\u20137979 (2021)","key":"29_CR26","DOI":"10.1109\/ICCV48922.2021.00787"},{"doi-asserted-by":"crossref","unstructured":"Yang, A., Miech, A., Sivic, J., Laptev, I., Schmid, C.: Tubedetr: Spatio-temporal video grounding with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16442\u201316453 (2022)","key":"29_CR27","DOI":"10.1109\/CVPR52688.2022.01595"},{"doi-asserted-by":"crossref","unstructured":"Yao, T., Mei, T., Rui, Y.: Highlight detection with pairwise deep ranking for first-person video summarization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 982\u2013990 (2016)","key":"29_CR28","DOI":"10.1109\/CVPR.2016.112"},{"doi-asserted-by":"crossref","unstructured":"Ye, Q., Shen, X., Gao, Y., Wang, Z., Bi, Q., Li, P., Yang, G.: Temporal cue guided video highlight detection with low-rank audio-visual fusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7950\u20137959 (2021)","key":"29_CR29","DOI":"10.1109\/ICCV48922.2021.00785"},{"doi-asserted-by":"crossref","unstructured":"Zeng, R., Xu, H., Huang, W., Chen, P., Tan, M., Gan, C.: Dense regression network for video grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10287\u201310296 (2020)","key":"29_CR30","DOI":"10.1109\/CVPR42600.2020.01030"},{"doi-asserted-by":"crossref","unstructured":"Zhang, H., Sun, A., Jing, W., Zhou, J.T.: Span-based localizing network for natural language video localization (2020). arXiv:2004.13931","key":"29_CR31","DOI":"10.18653\/v1\/2020.acl-main.585"},{"doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2d temporal adjacent networks for moment localization with natural language. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 12870\u201312877 (2020)","key":"29_CR32","DOI":"10.1609\/aaai.v34i07.6984"},{"doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2D temporal adjacent networks for moment localization with natural language. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 12870\u201312877 (2020)","key":"29_CR33","DOI":"10.1609\/aaai.v34i07.6984"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhao, Z., Zhao, Y., Wang, Q., Liu, H., Gao, L.: Where does it exist: Spatio-temporal video grounding for multi-form sentences. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10668\u201310677 (2020)","key":"29_CR34","DOI":"10.1109\/CVPR42600.2020.01068"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8792-0_29","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,8]],"date-time":"2024-11-08T07:11:29Z","timestamp":1731049889000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8792-0_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,9]]},"ISBN":["9789819787913","9789819787920"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8792-0_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,9]]},"assertion":[{"value":"9 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}