{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:38:50Z","timestamp":1782833930093,"version":"3.54.5"},"publisher-location":"Cham","reference-count":55,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726637","type":"print"},{"value":"9783031726644","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72664-4_20","type":"book-chapter","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T17:02:04Z","timestamp":1729875724000},"page":"352-369","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["RGNet: A Unified Clip Retrieval and\u00a0Grounding Network for\u00a0Long Videos"],"prefix":"10.1007","author":[{"given":"Tanveer","family":"Hannan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Md Mohaiminul","family":"Islam","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Thomas","family":"Seidl","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gedas","family":"Bertasius","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,26]]},"reference":[{"key":"20_CR1","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. In: ICCV, pp. 1728\u20131738 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Barrios, W., Soldan, M., Ceballos-Arroyo, A.M., Heilbron, F.C., Ghanem, B.: Localizing moments in long video via multimodal guidance. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13667\u201313678 (2023)","DOI":"10.1109\/ICCV51070.2023.01257"},{"key":"20_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"key":"20_CR4","doi-asserted-by":"crossref","unstructured":"Chen, S., Zhao, Y., Jin, Q., Wu, Q.: Fine-grained video-text retrieval with hierarchical graph reasoning. In: CVPR, pp. 10638\u201310647 (2020)","DOI":"10.1109\/CVPR42600.2020.01065"},{"key":"20_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Z., Jiang, X., Xu, X., Cao, Z., Mo, Y., Shen, H.T.: Joint searching and grounding: multi-granularity video content retrieval. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 975\u2013983 (2023)","DOI":"10.1145\/3581783.3612349"},{"key":"20_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, F., Wang, X., Lei, J., Crandall, D., Bansal, M., Bertasius, G.: VindLU: a recipe for effective video-and-language pretraining. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2023","DOI":"10.1109\/CVPR52729.2023.01034"},{"key":"20_CR7","doi-asserted-by":"crossref","unstructured":"Dong, J., et al.: Dual encoding for video retrieval by text. IEEE TPAMI, 4065\u20134080 (2021)","DOI":"10.1109\/TPAMI.2021.3059295"},{"key":"20_CR8","doi-asserted-by":"crossref","unstructured":"Fang, B., et\u00a0al.: UATVR: uncertainty-adaptive text-video retrieval. arXiv preprint arXiv:2301.06309 (2023)","DOI":"10.1109\/ICCV51070.2023.01262"},{"key":"20_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"214","DOI":"10.1007\/978-3-030-58548-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"V Gabeur","year":"2020","unstructured":"Gabeur, V., Sun, C., Alahari, K., Schmid, C.: Multi-modal transformer for video retrieval. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12349, pp. 214\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58548-8_13"},{"key":"20_CR10","unstructured":"Gao, Z., Liu, J., Sun, W., Chen, S., Chang, D., Zhao, L.: CLIP2TV: align, match and distill for video-text retrieval. arXiv preprint arXiv:2111.05610 (2021)"},{"key":"20_CR11","doi-asserted-by":"crossref","unstructured":"Ge, Y., et al.: Bridging video-text retrieval with multiple choice questions. In: CVPR, pp. 16167\u201316176 (2022)","DOI":"10.1109\/CVPR52688.2022.01569"},{"key":"20_CR12","unstructured":"Glorot, X., Bengio, Y.: Understanding the difficulty of training deep feedforward neural networks. In: Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics, pp. 249\u2013256. JMLR Workshop and Conference Proceedings (2010)"},{"key":"20_CR13","doi-asserted-by":"crossref","unstructured":"Gorti, S.K., et al.: X-Pool: cross-modal language-video attention for text-video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5006\u20135015 (2022)","DOI":"10.1109\/CVPR52688.2022.00495"},{"key":"20_CR14","unstructured":"Grauman, K., et\u00a0al.: Ego4D: around the world in 3,000 hours of egocentric video. In: CVPR, pp. 18995\u201319012 (2022)"},{"key":"20_CR15","unstructured":"Hannan, T., et al.: GRAtt-VIS: gated residual attention for auto rectifying video instance segmentation. arXiv preprint arXiv:2305.17096 (2023)"},{"key":"20_CR16","doi-asserted-by":"crossref","unstructured":"Hou, Z., et al.: CONE: an efficient coarse-to-fine alignment framework for long video temporal grounding. arXiv preprint arXiv:2209.10918 (2022)","DOI":"10.18653\/v1\/2023.acl-long.445"},{"key":"20_CR17","unstructured":"Jang, E., Gu, S., Poole, B.: Categorical reparameterization with Gumbel-Softmax. arXiv preprint arXiv:1611.01144 (2016)"},{"key":"20_CR18","doi-asserted-by":"crossref","unstructured":"Jang, J., Park, J., Kim, J., Kwon, H., Sohn, K.: Knowing where to focus: event-aware transformer for video grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13846\u201313856 (2023)","DOI":"10.1109\/ICCV51070.2023.01273"},{"key":"20_CR19","unstructured":"Jiang, H., et al.: Cross-modal adapter for text-video retrieval. arXiv preprint arXiv:2211.09623 (2022)"},{"key":"20_CR20","unstructured":"Jin, P., et al.: Expectation-maximization contrastive learning for compact video-and-language representations. In: NeurIPS (2022)"},{"key":"20_CR21","unstructured":"Kiros, R., Salakhutdinov, R., Zemel, R.S.: Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539 (2014)"},{"key":"20_CR22","unstructured":"Lei, J., Berg, T.L., Bansal, M.: Detecting moments and highlights in videos via natural language queries. In: Advances in Neural Information Processing Systems, vol. 34, pp. 11846\u201311858 (2021)"},{"key":"20_CR23","doi-asserted-by":"crossref","unstructured":"Lei, J., et al.: Less is more: ClipBERT for video-and-language learning via sparse sampling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7331\u20137341 (2021)","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"20_CR24","unstructured":"Lin, K.Q., et\u00a0al.: Egocentric video-language pretraining. arXiv preprint arXiv:2206.01670 (2022)"},{"key":"20_CR25","doi-asserted-by":"crossref","unstructured":"Lin, K.Q., et al.: UniVTG: towards unified video-language temporal grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2794\u20132804 (2023)","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"20_CR26","unstructured":"Liu, N., Wang, X., Li, X., Yang, Y., Zhuang, Y.: RELER@ZJU-Alibaba submission to the Ego4D natural language queries challenge 2022. arXiv preprint arXiv:2207.00383 (2022)"},{"key":"20_CR27","unstructured":"Liu, S., et al.: DAB-DETR: dynamic anchor boxes are better queries for DETR. arXiv preprint arXiv:2201.12329 (2022)"},{"key":"20_CR28","doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, S., Wu, Y., Chen, C.W., Shan, Y., Qie, X.: UMT: unified multi-modal transformers for joint video moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3042\u20133051 (2022)","DOI":"10.1109\/CVPR52688.2022.00305"},{"key":"20_CR29","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1007\/978-3-031-19781-9_19","volume-title":"ECCV 2022","author":"Y Liu","year":"2022","unstructured":"Liu, Y., Xiong, P., Xu, L., Cao, S., Jin, Q.: TS2-Net: token shift and selection transformer for text-video retrieval. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13674, pp. 319\u2013335. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19781-9_19"},{"key":"20_CR30","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"20_CR31","unstructured":"Luo, H., et al.: UniVL: a unified video and language pre-training model for multimodal understanding and generation. arXiv preprint arXiv:2002.06353 (2020)"},{"key":"20_CR32","doi-asserted-by":"crossref","unstructured":"Ma, Y., Xu, G., Sun, X., Yan, M., Zhang, J., Ji, R.: X-CLIP: end-to-end multi-grained contrastive learning for video-text retrieval. In: ACM MM, pp. 638\u2013647 (2022)","DOI":"10.1145\/3503161.3547910"},{"key":"20_CR33","doi-asserted-by":"crossref","unstructured":"Meng, D., et al.: Conditional DETR for fast training convergence. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3651\u20133660 (2021)","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"20_CR34","doi-asserted-by":"crossref","unstructured":"Moon, W., Hyun, S., Park, S., Park, D., Heo, J.P.: Query-dependent video representation for moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23023\u201323033 (2023)","DOI":"10.1109\/CVPR52729.2023.02205"},{"key":"20_CR35","unstructured":"Oord, A.V.D., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"20_CR36","doi-asserted-by":"crossref","unstructured":"Pan, Y., et al.: Scanning only once: an end-to-end framework for fast temporal grounding in long videos. arXiv preprint arXiv:2303.08345 (2023)","DOI":"10.1109\/ICCV51070.2023.01266"},{"key":"20_CR37","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision, pp. 8748\u20138763 (2021)"},{"key":"20_CR38","doi-asserted-by":"crossref","unstructured":"Ramakrishnan, S.K., Al-Halah, Z., Grauman, K.: NaQ: leveraging narrations as queries to supervise episodic memory. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6694\u20136703 (2023)","DOI":"10.1109\/CVPR52729.2023.00647"},{"key":"20_CR39","doi-asserted-by":"crossref","unstructured":"Soldan, M., et al.: MAD: a scalable dataset for language grounding in videos from movie audio descriptions. In: CVPR, pp. 5026\u20135035 (2022)","DOI":"10.1109\/CVPR52688.2022.00497"},{"key":"20_CR40","doi-asserted-by":"crossref","unstructured":"Soldan, M., Xu, M., Qu, S., Tegner, J., Ghanem, B.: VLG-Net: video-language graph matching network for video grounding. In: ICCV, pp. 3224\u20133234 (2021)","DOI":"10.1109\/ICCVW54120.2021.00361"},{"key":"20_CR41","unstructured":"Wang, A.J., et al.: All in one: exploring unified video-language pre-training. arXiv preprint arXiv:2203.07303 (2022)"},{"key":"20_CR42","unstructured":"Wang, Q., Zhang, Y., Zheng, Y., Pan, P., Hua, X.S.: Disentangled representation learning for text-video retrieval. arXiv preprint arXiv:2203.07111 (2022)"},{"key":"20_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Z., Sung, Y.L., Cheng, F., Bertasius, G., Bansal, M.: Unified coarse-to-fine alignment for video-text retrieval. In: The IEEE International Conference on Computer Vision (ICCV), October 2023","DOI":"10.1109\/ICCV51070.2023.00264"},{"key":"20_CR44","doi-asserted-by":"crossref","unstructured":"Xu, H., et al.: VLM: task-agnostic video-language model pre-training for video understanding. ACLliu2021hit (2021)","DOI":"10.18653\/v1\/2021.findings-acl.370"},{"key":"20_CR45","doi-asserted-by":"crossref","unstructured":"Xu, H., et al.: VideoCLIP: contrastive pre-training for zero-shot video-text understanding. In: EMNLP, pp. 6787\u20136800 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"20_CR46","doi-asserted-by":"crossref","unstructured":"Xue, H., et al.: Advancing high-resolution video-language representation with large-scale video transcriptions. In: CVPR, pp. 5036\u20135045 (2022)","DOI":"10.1109\/CVPR52688.2022.00498"},{"key":"20_CR47","unstructured":"Xue, H., et al.: CLIP-ViP: adapting pre-trained image-text model to video-language representation alignment. In: ICLR (2023)"},{"key":"20_CR48","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"487","DOI":"10.1007\/978-3-030-01234-2_29","volume-title":"Computer Vision \u2013 ECCV 2018","author":"Y Yu","year":"2018","unstructured":"Yu, Y., Kim, J., Kim, G.: A joint sequence fusion model for video question answering and retrieval. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11211, pp. 487\u2013503. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_29"},{"key":"20_CR49","doi-asserted-by":"crossref","unstructured":"Yu, Y., Ko, H., Choi, J., Kim, G.: End-to-end concept word detection for video captioning, retrieval, and question answering. In: CVPR, pp. 3165\u20133173 (2017)","DOI":"10.1109\/CVPR.2017.347"},{"key":"20_CR50","unstructured":"Zhang, B., et al.: Multimodal video adapter for parameter efficient video text retrieval. arXiv preprint arXiv:2301.07868 (2023)"},{"key":"20_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, C., Gupta, A., Zisserman, A.: Helping hands: an object-aware ego-centric video recognition model. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13901\u201313912 (2023)","DOI":"10.1109\/ICCV51070.2023.01278"},{"key":"20_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, H., Sun, A., Jing, W., Zhou, J.T.: Span-based localizing network for natural language video localization. arXiv preprint arXiv:2004.13931 (2020)","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"20_CR53","unstructured":"Zhang, H., Sun, A., Jing, W., Zhou, J.T.: The elements of temporal sentence grounding in videos: a survey and future directions. arXiv preprint arXiv:2201.080711(2) (2022)"},{"key":"20_CR54","doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2D temporal adjacent networks for moment localization with natural language. In: AAAI, pp. 12870\u201312877 (2020)","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"20_CR55","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72664-4_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T17:07:40Z","timestamp":1729876060000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72664-4_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,26]]},"ISBN":["9783031726637","9783031726644"],"references-count":55,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72664-4_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,26]]},"assertion":[{"value":"26 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}