{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:16:08Z","timestamp":1784178968997,"version":"3.55.0"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031731945","type":"print"},{"value":"9783031731952","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:00:00Z","timestamp":1732665600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:00:00Z","timestamp":1732665600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73195-2_24","type":"book-chapter","created":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T09:37:49Z","timestamp":1732613869000},"page":"412-429","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Weakly-Supervised Spatio-Temporal Video Grounding with\u00a0Variational Cross-Modal Alignment"],"prefix":"10.1007","author":[{"given":"Yang","family":"Jin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yadong","family":"Mu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,27]]},"reference":[{"key":"24_CR1","doi-asserted-by":"crossref","unstructured":"Arbelle, A., et\u00a0al.: Detector-free weakly supervised grounding by separation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1801\u20131812 (2021)","DOI":"10.1109\/ICCV48922.2021.00182"},{"key":"24_CR2","doi-asserted-by":"crossref","unstructured":"Bao, P., Shao, Z., Yang, W., Ng, B.P., Kot, A.C.: E3m: zero-shot spatio-temporal video grounding with expectation-maximization multimodal modulation. In: ECCV (2024)","DOI":"10.1007\/978-3-031-73010-8_14"},{"issue":"453\u2013464","key":"24_CR3","first-page":"210","volume":"7","author":"J Bernardo","year":"2003","unstructured":"Bernardo, J., et al.: The variational bayesian em algorithm for incomplete data: with application to scoring graphical model structures. Bayesian Stat. 7(453\u2013464), 210 (2003)","journal-title":"Bayesian Stat."},{"issue":"3","key":"24_CR4","first-page":"179","volume":"24","author":"J Besag","year":"1975","unstructured":"Besag, J.: Statistical analysis of non-lattice data. J. Roy. Stat. Soc. Ser. D (Stat.) 24(3), 179\u2013195 (1975)","journal-title":"J. Roy. Stat. Soc. Ser. D (Stat.)"},{"issue":"518","key":"24_CR5","doi-asserted-by":"publisher","first-page":"859","DOI":"10.1080\/01621459.2017.1285773","volume":"112","author":"DM Blei","year":"2017","unstructured":"Blei, D.M., Kucukelbir, A., McAuliffe, J.D.: Variational inference: a review for statisticians. J. Am. Stat. Assoc. 112(518), 859\u2013877 (2017)","journal-title":"J. Am. Stat. Assoc."},{"key":"24_CR6","doi-asserted-by":"crossref","unstructured":"Chen, J., Chen, X., Ma, L., Jie, Z., Chua, T.S.: Temporally grounding natural sentence in video. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 162\u2013171 (2018)","DOI":"10.18653\/v1\/D18-1015"},{"key":"24_CR7","doi-asserted-by":"crossref","unstructured":"Chen, J., Ma, L., Chen, X., Jie, Z., Luo, J.: Localizing natural language in videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 8175\u20138182 (2019)","DOI":"10.1609\/aaai.v33i01.33018175"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Chen, J., Bao, W., Kong, Y.: Activity-driven weakly-supervised spatio-temporal grounding from untrimmed videos. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 3789\u20133797 (2020)","DOI":"10.1145\/3394171.3413614"},{"key":"24_CR9","doi-asserted-by":"crossref","unstructured":"Chen, Z., Ma, L., Luo, W., Wong, K.Y.K.: Weakly-supervised spatio-temporally grounding natural sentence in video. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 1884\u20131894 (2019)","DOI":"10.18653\/v1\/P19-1183"},{"issue":"1\u201368","key":"24_CR10","first-page":"1","volume":"2000","author":"RT Collins","year":"2000","unstructured":"Collins, R.T., et al.: A system for video surveillance and monitoring. VSAM Final Rep. 2000(1\u201368), 1 (2000)","journal-title":"VSAM Final Rep."},{"key":"24_CR11","doi-asserted-by":"crossref","unstructured":"Datta, S., Sikka, K., Roy, A., Ahuja, K., Parikh, D., Divakaran, A.: Align2ground: weakly supervised phrase grounding guided by image-caption alignment. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2601\u20132610 (2019)","DOI":"10.1109\/ICCV.2019.00269"},{"key":"24_CR12","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: Tall: Temporal activity localization via language query. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5267\u20135275 (2017)","DOI":"10.1109\/ICCV.2017.563"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Malik, J.: Finding action tubes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 759\u2013768 (2015)","DOI":"10.1109\/CVPR.2015.7298676"},{"key":"24_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"24_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"24_CR16","first-page":"29192","volume":"35","author":"Y Jin","year":"2022","unstructured":"Jin, Y., Yuan, Z., Mu, Y., et al.: Embracing consistency: a one-stage approach for spatio-temporal video grounding. Adv. Neural. Inf. Process. Syst. 35, 29192\u201329204 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"24_CR17","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Bansal, M., Berg, T.L.: Tvqa: localized, compositional video question answering. arXiv preprint arXiv:1809.01696 (2018)","DOI":"10.18653\/v1\/D18-1167"},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: Winner: weakly-supervised hierarchical decomposition and alignment for spatio-temporal video grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23090\u201323099 (2023)","DOI":"10.1109\/CVPR52729.2023.02211"},{"key":"24_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"24_CR21","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wan, B., Ma, L., He, X.: Relation-aware instance refinement for weakly supervised visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5612\u20135621 (2021)","DOI":"10.1109\/CVPR46437.2021.00556"},{"key":"24_CR22","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"24_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"729","DOI":"10.1007\/978-3-030-58526-6_43","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Z Luo","year":"2020","unstructured":"Luo, Z., et al.: Weakly-supervised action localization with expectation-maximization multi-instance learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12374, pp. 729\u2013745. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58526-6_43"},{"key":"24_CR24","doi-asserted-by":"crossref","unstructured":"Manning, C.D., Surdeanu, M., Bauer, J., Finkel, J.R., Bethard, S., McClosky, D.: The stanford corenlp natural language processing toolkit. In: Proceedings of 52nd Annual Meeting of the Association for Computational Linguistics: System Demonstrations, pp. 55\u201360 (2014)","DOI":"10.3115\/v1\/P14-5010"},{"key":"24_CR25","doi-asserted-by":"crossref","unstructured":"Mun, J., Cho, M., Han, B.: Local-global video-text interactions for temporal grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10810\u201310819 (2020)","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"24_CR26","doi-asserted-by":"crossref","unstructured":"Neal, R.M., Hinton, G.E.: A view of the em algorithm that justifies incremental, sparse, and other variants. In: Learning in Graphical Models, pp. 355\u2013368 (1998)","DOI":"10.1007\/978-94-011-5014-9_12"},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: Glove: global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"24_CR28","unstructured":"Qu, M., Tang, J.: Probabilistic logic neural networks for reasoning. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"24_CR29","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. Adv. Neural Inf. Process. Syst. 28 (2015)"},{"key":"24_CR30","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1007\/s10994-006-5833-1","volume":"62","author":"M Richardson","year":"2006","unstructured":"Richardson, M., Domingos, P.: Markov logic networks. Mach. Learn. 62, 107\u2013136 (2006)","journal-title":"Mach. Learn."},{"key":"24_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"817","DOI":"10.1007\/978-3-319-46448-0_49","volume-title":"Computer Vision \u2013 ECCV 2016","author":"A Rohrbach","year":"2016","unstructured":"Rohrbach, A., Rohrbach, M., Hu, R., Darrell, T., Schiele, B.: Grounding of textual phrases in images by reconstruction. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 817\u2013834. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_49"},{"key":"24_CR32","doi-asserted-by":"crossref","unstructured":"Shi, J., Xu, J., Gong, B., Xu, C.: Not all frames are equal: weakly-supervised video grounding with contextual similarity and visual clustering losses. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10444\u201310452 (2019)","DOI":"10.1109\/CVPR.2019.01069"},{"key":"24_CR33","doi-asserted-by":"crossref","unstructured":"Su, R., Yu, Q., Xu, D.: Stvgbert: a visual-linguistic transformer based framework for spatio-temporal video grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1533\u20131542 (2021)","DOI":"10.1109\/ICCV48922.2021.00156"},{"key":"24_CR34","doi-asserted-by":"crossref","unstructured":"Tang, Z., et al.: Human-centric spatio-temporal video grounding with visual transformers. IEEE Trans. Circuits Syst. Video Technol. (2021)","DOI":"10.1109\/TCSVT.2021.3085907"},{"key":"24_CR35","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"24_CR36","doi-asserted-by":"crossref","unstructured":"Wang, L., Huang, J., Li, Y., Xu, K., Yang, Z., Yu, D.: Improving weakly supervised visual grounding by contrastive knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14090\u201314100 (2021)","DOI":"10.1109\/CVPR46437.2021.01387"},{"key":"24_CR37","doi-asserted-by":"crossref","unstructured":"Xiao, F., Sigal, L., Jae\u00a0Lee, Y.: Weakly-supervised visual grounding of phrases with linguistic structures. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5945\u20135954 (2017)","DOI":"10.1109\/CVPR.2017.558"},{"key":"24_CR38","doi-asserted-by":"crossref","unstructured":"Yang, A., Miech, A., Sivic, J., Laptev, I., Schmid, C.: Tubedetr: spatio-temporal video grounding with transformers. arXiv preprint arXiv:2203.16434 (2022)","DOI":"10.1109\/CVPR52688.2022.01595"},{"key":"24_CR39","doi-asserted-by":"publisher","first-page":"3252","DOI":"10.1109\/TIP.2021.3058614","volume":"30","author":"W Yang","year":"2021","unstructured":"Yang, W., Zhang, T., Zhang, Y., Wu, F.: Local correspondence network for weakly supervised temporal sentence grounding. IEEE Trans. Image Process. 30, 3252\u20133262 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"24_CR40","doi-asserted-by":"crossref","unstructured":"Zeng, R., Xu, H., Huang, W., Chen, P., Tan, M., Gan, C.: Dense regression network for video grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10287\u201310296 (2020)","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"24_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Lin, Z., Zhao, Z., Zhu, J., He, X.: Regularized two-branch proposal networks for weakly-supervised moment retrieval in videos. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4098\u20134106 (2020)","DOI":"10.1145\/3394171.3413967"},{"key":"24_CR42","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhao, Z., Lin, Z., Huai, B., Yuan, J.: Object-aware multi-branch relation networks for spatio-temporal video grounding. In: Proceedings of the Twenty-Ninth International Conference on International Joint Conferences on Artificial Intelligence, pp. 1069\u20131075 (2021)","DOI":"10.24963\/ijcai.2020\/149"},{"key":"24_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhao, Z., Zhao, Y., Wang, Q., Liu, H., Gao, L.: Where does it exist: spatio-temporal video grounding for multi-form sentences. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10668\u201310677 (2020)","DOI":"10.1109\/CVPR42600.2020.01068"},{"key":"24_CR44","doi-asserted-by":"crossref","unstructured":"Zhao, F., Li, J., Zhao, J., Feng, J.: Weakly supervised phrase localization with multi-scale anchored transformer network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5696\u20135705 (2018)","DOI":"10.1109\/CVPR.2018.00597"},{"key":"24_CR45","doi-asserted-by":"crossref","unstructured":"Zheng, M., Huang, Y., Chen, Q., Peng, Y., Liu, Y.: Weakly supervised temporal sentence grounding with gaussian-based contrastive proposal learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15555\u201315564 (2022)","DOI":"10.1109\/CVPR52688.2022.01511"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73195-2_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T10:14:17Z","timestamp":1732616057000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73195-2_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,27]]},"ISBN":["9783031731945","9783031731952"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73195-2_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,27]]},"assertion":[{"value":"27 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}