{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T07:10:38Z","timestamp":1767337838182,"version":"3.40.3"},"publisher-location":"Cham","reference-count":87,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031197802"},{"type":"electronic","value":"9783031197819"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-19781-9_33","type":"book-chapter","created":{"date-parts":[[2022,10,22]],"date-time":"2022-10-22T12:12:59Z","timestamp":1666440779000},"page":"567-587","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Text-Based Temporal Localization of\u00a0Novel Events"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7749-8594","authenticated-orcid":false,"given":"Sudipta","family":"Paul","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3611-4141","authenticated-orcid":false,"given":"Niluthpol Chowdhury","family":"Mithun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6690-9725","authenticated-orcid":false,"given":"Amit K.","family":"Roy-Chowdhury","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,23]]},"reference":[{"issue":"7","key":"33_CR1","doi-asserted-by":"publisher","first-page":"1425","DOI":"10.1109\/TPAMI.2015.2487986","volume":"38","author":"Z Akata","year":"2015","unstructured":"Akata, Z., Perronnin, F., Harchaoui, Z., Schmid, C.: Label-embedding for image classification. IEEE Trans. Pattern Anal. Mach. Intell. 38(7), 1425\u20131438 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"33_CR2","doi-asserted-by":"crossref","unstructured":"Akata, Z., Reed, S., Walter, D., Lee, H., Schiele, B.: Evaluation of output embeddings for fine-grained image classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2927\u20132936 (2015)","DOI":"10.1109\/CVPR.2015.7298911"},{"key":"33_CR3","doi-asserted-by":"crossref","unstructured":"Anne Hendricks, L., Wang, O., Shechtman, E., Sivic, J., Darrell, T., Russell, B.: Localizing moments in video with natural language. In: Proceedings of the IEEE international conference on computer vision. pp. 5803\u20135812 (2017)","DOI":"10.1109\/ICCV.2017.618"},{"key":"33_CR4","unstructured":"Cer, D., et al.: Universal sentence encoder. arXiv preprint arXiv:1803.11175 (2018)"},{"key":"33_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Chen, X., Ma, L., Jie, Z., Chua, T.S.: Temporally grounding natural sentence in video. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 162\u2013171 (2018)","DOI":"10.18653\/v1\/D18-1015"},{"key":"33_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"333","DOI":"10.1007\/978-3-030-58548-8_20","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Chen","year":"2020","unstructured":"Chen, S., Jiang, W., Liu, W., Jiang, Y.-G.: Learning modality interaction for temporal sentence localization and event captioning in videos. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12349, pp. 333\u2013351. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58548-8_20"},{"key":"33_CR7","doi-asserted-by":"crossref","unstructured":"Chen, S., Jiang, Y.G.: Towards bridging event captioner and sentence localizer for weakly supervised dense event captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8425\u20138435 (2021)","DOI":"10.1109\/CVPR46437.2021.00832"},{"key":"33_CR8","doi-asserted-by":"crossref","unstructured":"Chi, J., Peng, Y.: Dual adversarial networks for zero-shot cross-media retrieval. In: IJCAI, pp. 663\u2013669 (2018)","DOI":"10.24963\/ijcai.2018\/92"},{"key":"33_CR9","doi-asserted-by":"crossref","unstructured":"Ding, X., et al.: Support-set based cross-supervision for video grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11573\u201311582 (2021)","DOI":"10.1109\/ICCV48922.2021.01137"},{"key":"33_CR10","doi-asserted-by":"crossref","unstructured":"Dong, J., et al.: Dual encoding for zero-example video retrieval. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9346\u20139355 (2019)","DOI":"10.1109\/CVPR.2019.00957"},{"key":"33_CR11","unstructured":"Frome, A., et al.: Devise: a deep visual-semantic embedding model. In: Advances in Neural Information Processing Systems (NIPS), pp. 2121\u20132129 (2013)"},{"key":"33_CR12","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: TALL: temporal activity localization via language query. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5267\u20135275 (2017)","DOI":"10.1109\/ICCV.2017.563"},{"key":"33_CR13","doi-asserted-by":"crossref","unstructured":"Gao, J., Xu, C.: Fast video moment retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1523\u20131532 (2021)","DOI":"10.1109\/ICCV48922.2021.00155"},{"key":"33_CR14","doi-asserted-by":"crossref","unstructured":"Ge, R., Gao, J., Chen, K., Nevatia, R.: MAC: mining activity concepts for language-based temporal localization. In: 2019 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 245\u2013253. IEEE (2019)","DOI":"10.1109\/WACV.2019.00032"},{"key":"33_CR15","unstructured":"Ghosh, S., Agarwal, A., Parekh, Z., Hauptmann, A.: ExCl: extractive clip localization using natural language descriptions. arXiv preprint arXiv:1904.02755 (2019)"},{"key":"33_CR16","unstructured":"Hahn, M., Kadav, A., Rehg, J.M., Graf, H.P.: Tripping through time: efficient localization of activities in videos. arXiv preprint arXiv:1904.09936 (2019)"},{"key":"33_CR17","doi-asserted-by":"crossref","unstructured":"He, D., Zhao, X., Huang, J., Li, F., Liu, X., Wen, S.: Read, watch, and move: reinforcement learning for temporally grounding natural language descriptions in videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 8393\u20138400 (2019)","DOI":"10.1609\/aaai.v33i01.33018393"},{"key":"33_CR18","doi-asserted-by":"crossref","unstructured":"Hendricks, L.A., Wang, O., Shechtman, E., Sivic, J., Darrell, T., Russell, B.: Localizing moments in video with temporal language. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 1380\u20131390 (2018)","DOI":"10.18653\/v1\/D18-1168"},{"issue":"8","key":"33_CR19","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"33_CR20","unstructured":"Honnibal, M., Montani, I.: spaCy 2: natural language understanding with bloom embeddings. convolutional neural networks and incremental parsing (to appear 2017)"},{"key":"33_CR21","doi-asserted-by":"crossref","unstructured":"Huang, J., Liu, Y., Gong, S., Jin, H.: Cross-sentence temporal and semantic relations in video activity localisation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7199\u20137208 (2021)","DOI":"10.1109\/ICCV48922.2021.00711"},{"key":"33_CR22","doi-asserted-by":"crossref","unstructured":"Jiang, B., Huang, X., Yang, C., Yuan, J.: Cross-modal video moment retrieval with spatial and language-temporal attention. In: Proceedings of the 2019 on International Conference on Multimedia Retrieval., pp. 217\u2013225 (2019)","DOI":"10.1145\/3323873.3325019"},{"key":"33_CR23","unstructured":"Kingma, D., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"33_CR24","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., Niebles, J.C.: Dense-captioning events in videos. In: International Conference on Computer Vision (ICCV) (2017)","DOI":"10.1109\/ICCV.2017.83"},{"issue":"3","key":"33_CR25","doi-asserted-by":"publisher","first-page":"453","DOI":"10.1109\/TPAMI.2013.140","volume":"36","author":"CH Lampert","year":"2013","unstructured":"Lampert, C.H., Nickisch, H., Harmeling, S.: Attribute-based classification for zero-shot visual object categorization. IEEE Trans. Pattern Anal. Mach. Intell. 36(3), 453\u2013465 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"33_CR26","doi-asserted-by":"crossref","unstructured":"Li, J., Jing, M., Lu, K., Ding, Z., Zhu, L., Huang, Z.: Leveraging the invariant side of generative zero-shot learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7402\u20137411 (2019)","DOI":"10.1109\/CVPR.2019.00758"},{"key":"33_CR27","doi-asserted-by":"crossref","unstructured":"Lin, K., Xu, X., Gao, L., Wang, Z., Shen, H.T.: Learning cross-aligned latent embeddings for zero-shot cross-modal retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11515\u201311522 (2020)","DOI":"10.1609\/aaai.v34i07.6817"},{"key":"33_CR28","doi-asserted-by":"crossref","unstructured":"Lin, Z., Zhao, Z., Zhang, Z., Wang, Q., Liu, H.: Weakly-supervised video moment retrieval via semantic completion network. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11539\u201311546 (2020)","DOI":"10.1609\/aaai.v34i07.6820"},{"key":"33_CR29","doi-asserted-by":"publisher","first-page":"3750","DOI":"10.1109\/TIP.2020.2965987","volume":"29","author":"Z Lin","year":"2020","unstructured":"Lin, Z., Zhao, Z., Zhang, Z., Zhang, Z., Cai, D.: Moment retrieval via cross-modal interaction networks with query reconstruction. IEEE Trans. Image Process. 29, 3750\u20133762 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"33_CR30","doi-asserted-by":"crossref","unstructured":"Liu, B., Yeung, S., Chou, E., Huang, D.A., Fei-Fei, L., Carlos Niebles, J.: Temporal modular networks for retrieving complex compositional activities in videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 552\u2013568 (2018)","DOI":"10.1007\/978-3-030-01219-9_34"},{"key":"33_CR31","doi-asserted-by":"crossref","unstructured":"Liu, D., et al.: Context-aware biaffine localizing network for temporal sentence grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11235\u201311244 (2021)","DOI":"10.1109\/CVPR46437.2021.01108"},{"key":"33_CR32","doi-asserted-by":"crossref","unstructured":"Liu, D., Qu, X., Liu, X.Y., Dong, J., Zhou, P., Xu, Z.: Jointly cross-and self-modal graph attention network for query-based moment localization. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4070\u20134078 (2020)","DOI":"10.1145\/3394171.3414026"},{"key":"33_CR33","doi-asserted-by":"crossref","unstructured":"Liu, M., Wang, X., Nie, L., He, X., Chen, B., Chua, T.S.: Attentive moment retrieval in videos. In: The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval, pp. 15\u201324 (2018)","DOI":"10.1145\/3209978.3210003"},{"key":"33_CR34","doi-asserted-by":"crossref","unstructured":"Liu, M., Wang, X., Nie, L., Tian, Q., Chen, B., Chua, T.S.: Cross-modal moment localization in videos. In: Proceedings of the 26th ACM International Conference on Multimedia, pp. 843\u2013851 (2018)","DOI":"10.1145\/3240508.3240549"},{"key":"33_CR35","doi-asserted-by":"crossref","unstructured":"Mithun, N.C., Paul, S., Roy-Chowdhury, A.K.: Weakly supervised video moment retrieval from text queries. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), June 2019","DOI":"10.1109\/CVPR.2019.01186"},{"key":"33_CR36","doi-asserted-by":"crossref","unstructured":"Mun, J., Cho, M., Han, B.: Local-global video-text interactions for temporal grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10810\u201310819 (2020)","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"33_CR37","doi-asserted-by":"crossref","unstructured":"Nam, J., Ahn, D., Kang, D., Ha, S.J., Choi, J.: Zero-shot natural language video localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1470\u20131479, October 2021","DOI":"10.1109\/ICCV48922.2021.00150"},{"key":"33_CR38","doi-asserted-by":"crossref","unstructured":"Nan, G., et al.: Interventional video grounding with dual contrastive learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2765\u20132775 (2021)","DOI":"10.1109\/CVPR46437.2021.00279"},{"issue":"2","key":"33_CR39","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1109\/TIP.2018.2872916","volume":"28","author":"L Niu","year":"2018","unstructured":"Niu, L., Cai, J., Veeraraghavan, A., Zhang, L.: Zero-shot learning via category-specific visual-semantic mapping and label refinement. IEEE Trans. Image Process. 28(2), 965\u2013979 (2018)","journal-title":"IEEE Trans. Image Process."},{"key":"33_CR40","unstructured":"Norouzi, M., et al.: Zero-shot learning by convex combination of semantic embeddings. arXiv preprint arXiv:1312.5650 (2013)"},{"key":"33_CR41","doi-asserted-by":"crossref","unstructured":"Parikh, D., Grauman, K.: Relative attributes. In: 2011 International Conference on Computer Vision, pp. 503\u2013510. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126281"},{"key":"33_CR42","unstructured":"Patacchiola, M., Storkey, A.: Self-supervised relational reasoning for representation learning. arXiv preprint arXiv:2006.05849 (2020)"},{"key":"33_CR43","doi-asserted-by":"publisher","first-page":"8886","DOI":"10.1109\/TIP.2021.3120038","volume":"30","author":"S Paul","year":"2021","unstructured":"Paul, S., Mithun, N.C., Roy-Chowdhury, A.K.: Text-based localization of moments in a video corpus. IEEE Trans. Image Process. 30, 8886\u20138899 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"33_CR44","doi-asserted-by":"crossref","unstructured":"Paul, S., Torres, C., Chandrasekaran, S., Roy-Chowdhury, A.K.: Complex pairwise activity analysis via instance level evolution reasoning. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 2378\u20132382. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053473"},{"key":"33_CR45","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: Glove: global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"33_CR46","unstructured":"Raposo, D., Santoro, A., Barrett, D., Pascanu, R., Lillicrap, T., Battaglia, P.: Discovering objects and their relations from entangled scene representations. arXiv preprint arXiv:1702.05068 (2017)"},{"key":"33_CR47","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1162\/tacl_a_00207","volume":"1","author":"M Regneri","year":"2013","unstructured":"Regneri, M., et al.: Grounding action descriptions in videos. Trans. Assoc. Comput. Linguis. 1, 25\u201336 (2013)","journal-title":"Trans. Assoc. Comput. Linguis."},{"key":"33_CR48","unstructured":"Romera-Paredes, B., Torr, P.: An embarrassingly simple approach to zero-shot learning. In: International Conference on Machine Learning, pp. 2152\u20132161. PMLR (2015)"},{"key":"33_CR49","unstructured":"Santoro, A., et al.: A simple neural network module for relational reasoning. arXiv preprint arXiv:1706.01427 (2017)"},{"key":"33_CR50","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"510","DOI":"10.1007\/978-3-319-46448-0_31","volume-title":"Computer Vision \u2013 ECCV 2016","author":"GA Sigurdsson","year":"2016","unstructured":"Sigurdsson, G.A., Varol, G., Wang, X., Farhadi, A., Laptev, I., Gupta, A.: Hollywood in homes: crowdsourcing data collection for activity understanding. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 510\u2013526. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_31"},{"key":"33_CR51","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"33_CR52","unstructured":"Socher, R., Ganjoo, M., Sridhar, H., Bastani, O., Manning, C.D., Ng, A.Y.: Zero-shot learning through cross-modal transfer. arXiv preprint arXiv:1301.3666 (2013)"},{"key":"33_CR53","doi-asserted-by":"crossref","unstructured":"Soldan, M., Xu, M., Qu, S., Tegner, J., Ghanem, B.: VLG-Net: video-language graph matching network for video grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3224\u20133234 (2021)","DOI":"10.1109\/ICCVW54120.2021.00361"},{"key":"33_CR54","doi-asserted-by":"crossref","unstructured":"Sung, F., Yang, Y., Zhang, L., Xiang, T., Torr, P.H., Hospedales, T.M.: Learning to compare: relation network for few-shot learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1199\u20131208 (2018)","DOI":"10.1109\/CVPR.2018.00131"},{"key":"33_CR55","doi-asserted-by":"crossref","unstructured":"Tan, R., Xu, H., Saenko, K., Plummer, B.A.: Logan: Latent graph co-attention network for weakly-supervised video moment retrieval. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 2083\u20132092 (2021)","DOI":"10.1109\/WACV48630.2021.00213"},{"key":"33_CR56","doi-asserted-by":"crossref","unstructured":"Tang, H., Zhu, J., Gao, Z., Zhuo, T., Cheng, Z.: Attention feature matching for weakly-supervised video relocalization. In: Proceedings of the 2nd ACM International Conference on Multimedia in Asia, pp. 1\u20137 (2021)","DOI":"10.1145\/3444685.3446317"},{"key":"33_CR57","doi-asserted-by":"crossref","unstructured":"Tang, H., Zhu, J., Wang, L., Zheng, Q., Zhang, T.: Multi-level query interaction for temporal language grounding. IEEE Transactions on Intelligent Transportation Systems (2021)","DOI":"10.1109\/TITS.2021.3110713"},{"key":"33_CR58","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: International Conference on Computer Vision (ICCV), pp. 4489\u20134497. IEEE (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"33_CR59","doi-asserted-by":"crossref","unstructured":"Wang, H., Zha, Z.J., Chen, X., Xiong, Z., Luo, J.: Dual path interaction network for video moment localization. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4116\u20134124 (2020)","DOI":"10.1145\/3394171.3413975"},{"key":"33_CR60","doi-asserted-by":"crossref","unstructured":"Wang, H., Zha, Z.J., Li, L., Liu, D., Luo, J.: Structured multi-level interaction network for video moment localization via language query. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7026\u20137035 (2021)","DOI":"10.1109\/CVPR46437.2021.00695"},{"key":"33_CR61","doi-asserted-by":"crossref","unstructured":"Wang, Y., Deng, J., Zhou, W., Li, H.: Weakly supervised temporal adjacent network for language grounding. IEEE Trans. Multim. 24, 3276\u20133286 (2021)","DOI":"10.1109\/TMM.2021.3096087"},{"key":"33_CR62","doi-asserted-by":"crossref","unstructured":"Wu, A., Han, Y.: Multi-modal circulant fusion for video-to-language and backward. In: Proceedings of the 27th International Joint Conference on Artificial Intelligence, pp. 1029\u20131035 (2018)","DOI":"10.24963\/ijcai.2018\/143"},{"key":"33_CR63","doi-asserted-by":"crossref","unstructured":"Xian, Y., Akata, Z., Sharma, G., Nguyen, Q., Hein, M., Schiele, B.: Latent embeddings for zero-shot classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 69\u201377 (2016)","DOI":"10.1109\/CVPR.2016.15"},{"key":"33_CR64","doi-asserted-by":"crossref","unstructured":"Xian, Y., Schiele, B., Akata, Z.: Zero-shot learning-the good, the bad and the ugly. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4582\u20134591 (2017)","DOI":"10.1109\/CVPR.2017.328"},{"key":"33_CR65","doi-asserted-by":"crossref","unstructured":"Xiao, S., et al.: Boundary proposal network for two-stage natural language video localization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 2986\u20132994 (2021)","DOI":"10.1609\/aaai.v35i4.16406"},{"key":"33_CR66","doi-asserted-by":"crossref","unstructured":"Xu, H., He, K., Plummer, B.A., Sigal, L., Sclaroff, S., Saenko, K.: Multilevel language and vision integration for text-to-clip retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 9062\u20139069 (2019)","DOI":"10.1609\/aaai.v33i01.33019062"},{"key":"33_CR67","doi-asserted-by":"crossref","unstructured":"Xu, M., et al.: Boundary-sensitive pre-training for temporal localization in videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7220\u20137230 (2021)","DOI":"10.1109\/ICCV48922.2021.00713"},{"key":"33_CR68","doi-asserted-by":"crossref","unstructured":"Xu, X., Song, J., Lu, H., Yang, Y., Shen, F., Huang, Z.: Modal-adversarial semantic learning network for extendable cross-modal retrieval. In: Proceedings of the 2018 ACM on International Conference on Multimedia Retrieval, pp. 46\u201354 (2018)","DOI":"10.1145\/3206025.3206033"},{"key":"33_CR69","doi-asserted-by":"crossref","unstructured":"Xu, X., Hospedales, T., Gong, S.: Semantic embedding space for zero-shot action recognition. In: 2015 IEEE International Conference on Image Processing (ICIP), pp. 63\u201367. IEEE (2015)","DOI":"10.1109\/ICIP.2015.7350760"},{"key":"33_CR70","doi-asserted-by":"publisher","first-page":"3252","DOI":"10.1109\/TIP.2021.3058614","volume":"30","author":"W Yang","year":"2021","unstructured":"Yang, W., Zhang, T., Zhang, Y., Wu, F.: Local correspondence network for weakly supervised temporal sentence grounding. IEEE Trans. Image Process. 30, 3252\u20133262 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"33_CR71","unstructured":"Yuan, Y., Ma, L., Wang, J., Liu, W., Zhu, W.: Semantic conditioned dynamic modulation for temporal sentence grounding in videos. In: Advances in Neural Information Processing Systems, pp. 534\u2013544 (2019)"},{"key":"33_CR72","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Mei, T., Zhu, W.: To find where you talk: Temporal sentence localization in video with attention based location regression. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 9159\u20139166 (2019)","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"33_CR73","unstructured":"Zambaldi, V., et al.: Deep reinforcement learning with relational inductive biases. In: International Conference on Learning Representations (2018)"},{"key":"33_CR74","doi-asserted-by":"crossref","unstructured":"Zeng, R., Xu, H., Huang, W., Chen, P., Tan, M., Gan, C.: Dense regression network for video grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10287\u201310296 (2020)","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"33_CR75","doi-asserted-by":"crossref","unstructured":"Zhang, D., Dai, X., Wang, X., Wang, Y.F., Davis, L.S.: MAN: moment alignment network for natural language moment retrieval via iterative graph adjustment. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1247\u20131257 (2019)","DOI":"10.1109\/CVPR.2019.00134"},{"key":"33_CR76","doi-asserted-by":"crossref","unstructured":"Zhang, H., Sun, A., Jing, W., Zhen, L., Zhou, J.T., Goh, R.S.M.: Natural language video localization: a revisit in span-based question answering framework. IEEE Trans. Pattern Anal. Mach. Intell. (2021)","DOI":"10.1109\/TPAMI.2021.3060449"},{"key":"33_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, H., Sun, A., Jing, W., Zhou, J.T.: Span-based localizing network for natural language video localization. arXiv preprint arXiv:2004.13931 (2020)","DOI":"10.18653\/v1\/2020.acl-main.585"},{"issue":"1","key":"33_CR78","doi-asserted-by":"publisher","first-page":"506","DOI":"10.1109\/TIP.2018.2869696","volume":"28","author":"H Zhang","year":"2018","unstructured":"Zhang, H., Long, Y., Guan, Y., Shao, L.: Triple verification network for generalized zero-shot learning. IEEE Trans. Image Process. 28(1), 506\u2013517 (2018)","journal-title":"IEEE Trans. Image Process."},{"key":"33_CR79","doi-asserted-by":"crossref","unstructured":"Zhang, L., et al.: Zstad: zero-shot temporal activity detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), June 2020","DOI":"10.1109\/CVPR42600.2020.00096"},{"key":"33_CR80","doi-asserted-by":"crossref","unstructured":"Zhang, M., et al.: Multi-stage aggregated transformer network for temporal language localization in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12669\u201312678 (2021)","DOI":"10.1109\/CVPR46437.2021.01248"},{"key":"33_CR81","doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2D temporal adjacent networks for moment localization with natural language. arXiv preprint arXiv:1912.03590 (2019)","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"33_CR82","doi-asserted-by":"crossref","unstructured":"Zhang, S., Su, J., Luo, J.: Exploiting temporal relationships in video moment localization with natural language. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 1230\u20131238 (2019)","DOI":"10.1145\/3343031.3350879"},{"key":"33_CR83","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Lin, Z., Zhao, Z., Xiao, Z.: Cross-modal interaction networks for query-based moment retrieval in videos. In: Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 655\u2013664 (2019)","DOI":"10.1145\/3331184.3331235"},{"key":"33_CR84","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Zhao, Z., Zhang, Z., Lin, Z.: Cascaded prediction network via segment tree for temporal video grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4197\u20134206 (2021)","DOI":"10.1109\/CVPR46437.2021.00418"},{"key":"33_CR85","doi-asserted-by":"crossref","unstructured":"Zhou, B., Andonian, A., Oliva, A., Torralba, A.: Temporal relational reasoning in videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 803\u2013818 (2018)","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"33_CR86","doi-asserted-by":"crossref","unstructured":"Zhou, H., Zhang, C., Luo, Y., Chen, Y., Hu, C.: Embracing uncertainty: decoupling and de-bias for robust temporal grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8445\u20138454 (2021)","DOI":"10.1109\/CVPR46437.2021.00834"},{"key":"33_CR87","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Long, Y., Guan, Y., Newsam, S., Shao, L.: Towards universal representation for unseen action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9436\u20139445 (2018)","DOI":"10.1109\/CVPR.2018.00983"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-19781-9_33","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,12]],"date-time":"2024-03-12T16:43:15Z","timestamp":1710261795000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-19781-9_33"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031197802","9783031197819"],"references-count":87,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-19781-9_33","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"23 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}