{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:41:43Z","timestamp":1777653703282,"version":"3.51.4"},"publisher-location":"Cham","reference-count":127,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729942","type":"print"},{"value":"9783031729959","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72995-9_17","type":"book-chapter","created":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T19:15:44Z","timestamp":1732389344000},"page":"290-311","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Rethinking Weakly-Supervised Video Temporal Grounding From a\u00a0Game Perspective"],"prefix":"10.1007","author":[{"given":"Xiang","family":"Fang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeyu","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wanlong","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoye","family":"Qu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianfeng","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keke","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daizong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,24]]},"reference":[{"key":"17_CR1","doi-asserted-by":"crossref","unstructured":"Albarelli, A., Rodola, E., Torsello, A.: A game-theoretic approach to fine surface registration without initial motion estimation. In: 2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, pp. 430\u2013437. IEEE (2010)","DOI":"10.1109\/CVPR.2010.5540183"},{"key":"17_CR2","doi-asserted-by":"crossref","unstructured":"Anne\u00a0Hendricks, L., Wang, O., Shechtman, E., Sivic, J., Darrell, T., Russell, B.: Localizing moments in video with natural language. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5803\u20135812 (2017)","DOI":"10.1109\/ICCV.2017.618"},{"key":"17_CR3","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1007\/s10458-009-9078-9","volume":"20","author":"Y Bachrach","year":"2010","unstructured":"Bachrach, Y., Markakis, E., Resnick, E., Procaccia, A.D., Rosenschein, J.S., Saberi, A.: Approximating power indices: theoretical and empirical analysis. Auton. Agent. Multi-Agent Syst. 20, 105\u2013122 (2010)","journal-title":"Auton. Agent. Multi-Agent Syst."},{"key":"17_CR4","first-page":"317","volume":"19","author":"JF Banzhaf III","year":"1964","unstructured":"Banzhaf, J.F., III.: Weighted voting doesn\u2019t work: a mathematical analysis. Rutgers L. Rev. 19, 317 (1964)","journal-title":"Rutgers L. Rev."},{"key":"17_CR5","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo Vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"issue":"6","key":"17_CR6","first-page":"1","volume":"5","author":"G Chalkiadakis","year":"2011","unstructured":"Chalkiadakis, G., Elkind, E., Wooldridge, M.: Computational aspects of cooperative game theory. Syn. Lect. Artif. Intell. Mach. Learn. 5(6), 1\u2013168 (2011)","journal-title":"Syn. Lect. Artif. Intell. Mach. Learn."},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Chen, J., Luo, W., Zhang, W., Ma, L.: Explore inter-contrast between videos via composition for weakly supervised temporal sentence grounding. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 267\u2013275 (2022)","DOI":"10.1609\/aaai.v36i1.19902"},{"key":"17_CR8","doi-asserted-by":"crossref","unstructured":"Chen, J., Ma, L., Chen, X., Jie, Z., Luo, J.: Localizing natural language in videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 8175\u20138182 (2019)","DOI":"10.1609\/aaai.v33i01.33018175"},{"key":"17_CR9","doi-asserted-by":"crossref","unstructured":"Chen, L., et al.: Rethinking the bottom-up framework for query-based video localization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 10551\u201310558 (2020)","DOI":"10.1609\/aaai.v34i07.6627"},{"key":"17_CR10","doi-asserted-by":"crossref","unstructured":"Chen, Z., Ma, L., Luo, W., Tang, P., Wong, K.Y.K.: Look closer to ground better: weakly-supervised temporal grounding of sentence in video. arXiv preprint arXiv:2001.09308 (2020)","DOI":"10.18653\/v1\/P19-1183"},{"key":"17_CR11","doi-asserted-by":"crossref","unstructured":"Datta, A., Sen, S., Zick, Y.: Algorithmic transparency via quantitative input influence: theory and experiments with learning systems. In: 2016 IEEE Symposium on Security and Privacy, pp. 598\u2013617. IEEE (2016)","DOI":"10.1109\/SP.2016.42"},{"key":"17_CR12","doi-asserted-by":"publisher","unstructured":"Deng, S., Wen, J., Liu, C., Yan, K., Xu, G., Xu, Y.: Projective incomplete multi-view clustering. IEEE Trans. Neural Netw. Learn. Syst. 35(8), 1\u201313 (2023). https:\/\/doi.org\/10.1109\/TNNLS.2023.3242473","DOI":"10.1109\/TNNLS.2023.3242473"},{"key":"17_CR13","doi-asserted-by":"crossref","unstructured":"Dong, J., et al.: Partially relevant video retrieval. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 246\u2013257 (2022)","DOI":"10.1145\/3503161.3547976"},{"issue":"12","key":"17_CR14","doi-asserted-by":"publisher","first-page":"3377","DOI":"10.1109\/TMM.2018.2832602","volume":"20","author":"J Dong","year":"2018","unstructured":"Dong, J., Li, X., Snoek, C.G.: Predicting visual features from text for image and video caption retrieval. IEEE Trans. Multimedia 20(12), 3377\u20133388 (2018)","journal-title":"IEEE Trans. Multimedia"},{"issue":"8","key":"17_CR15","first-page":"4065","volume":"44","author":"J Dong","year":"2022","unstructured":"Dong, J., et al.: Dual encoding for video retrieval by text. IEEE Trans. Pattern Anal. Mach. Intell. 44(8), 4065\u20134080 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"17_CR16","doi-asserted-by":"crossref","unstructured":"Dong, J., et al.: From region to patch: attribute-aware foreground-background contrastive learning for fine-grained fashion retrieval. In: Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 1273\u20131282 (2023)","DOI":"10.1145\/3539618.3591690"},{"key":"17_CR17","doi-asserted-by":"crossref","unstructured":"Dong, J., Sun, S., Liu, Z., Chen, S., Liu, B., Wang, X.: Hierarchical contrast for unsupervised skeleton-based action representation learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 525\u2013533 (2023)","DOI":"10.1609\/aaai.v37i1.25127"},{"issue":"8","key":"17_CR18","doi-asserted-by":"publisher","first-page":"5680","DOI":"10.1109\/TCSVT.2022.3150959","volume":"32","author":"J Dong","year":"2022","unstructured":"Dong, J., et al.: Reading-strategy inspired visual representation learning for text-to-video retrieval. IEEE Trans. Circuits Syst. Video Technol. 32(8), 5680\u20135694 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"17_CR19","doi-asserted-by":"crossref","unstructured":"Donoser, M., Bischof, H.: Diffusion processes for retrieval revisited. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1320\u20131327 (2013)","DOI":"10.1109\/CVPR.2013.174"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Dowdall, J., Pavlidis, I.T., Tsiamyrtzis, P.: Coalitional tracking in facial infrared imaging and beyond. In: 2006 Conference on Computer Vision and Pattern Recognition Workshop, pp. 134\u2013134. IEEE (2006)","DOI":"10.1109\/CVPRW.2006.55"},{"key":"17_CR21","doi-asserted-by":"crossref","unstructured":"Fang, X., Easwaran, A., Genest, B.: Uncertainty-guided appearance-motion association network for out-of-distribution action detection. In: IEEE International Conference on Multimedia Information Processing and Retrieval (2024)","DOI":"10.1109\/MIPR62202.2024.00034"},{"key":"17_CR22","doi-asserted-by":"crossref","unstructured":"Fang, X., et al.: Not all inputs are valid: towards open-set video moment retrieval using language. In: Proceedings of the 32th ACM International Conference on Multimedia (2024)","DOI":"10.1145\/3664647.3680947"},{"key":"17_CR23","unstructured":"Fang, X., Hu, Y.: Double self-weighted multi-view clustering via adaptive view fusion. arXiv preprint arXiv:2011.10396 (2020)"},{"issue":"2","key":"17_CR24","doi-asserted-by":"publisher","first-page":"192","DOI":"10.1109\/TAI.2021.3116546","volume":"3","author":"X Fang","year":"2021","unstructured":"Fang, X., Hu, Y., Zhou, P., Wu, D.: ANIMC: a soft approach for autoweighted noisy and incomplete multiview clustering. IEEE Trans. Artif. Intell. 3(2), 192\u2013206 (2021)","journal-title":"IEEE Trans. Artif. Intell."},{"issue":"3","key":"17_CR25","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1109\/TAI.2021.3052425","volume":"1","author":"X Fang","year":"2020","unstructured":"Fang, X., Hu, Y., Zhou, P., Wu, D.O.: V3H: view variation and view heredity for incomplete multiview clustering. IEEE Trans. Artif. Intell. 1(3), 233\u2013247 (2020)","journal-title":"IEEE Trans. Artif. Intell."},{"issue":"4","key":"17_CR26","doi-asserted-by":"publisher","first-page":"913","DOI":"10.1109\/TETCI.2021.3077909","volume":"6","author":"X Fang","year":"2021","unstructured":"Fang, X., Hu, Y., Zhou, P., Wu, D.O.: Unbalanced incomplete multi-view clustering via the scheme of view evolution: weak views are meat; strong views do eat. IEEE Trans. Emerg. Top. Comput. Intell. 6(4), 913\u2013927 (2021)","journal-title":"IEEE Trans. Emerg. Top. Comput. Intell."},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Fang, X., et al.: Annotations are not all you need: a cross-modal knowledge transfer network for unsupervised temporal sentence grounding. In: Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 8721\u20138733 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.583"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Fang, X., et al.: Fewer steps, better performance: efficient cross-modal clip trimming for video moment retrieval using language. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 1735\u20131743 (2024)","DOI":"10.1609\/aaai.v38i2.27941"},{"key":"17_CR29","doi-asserted-by":"publisher","first-page":"7517","DOI":"10.1109\/TMM.2022.3222965","volume":"25","author":"X Fang","year":"2022","unstructured":"Fang, X., Liu, D., Zhou, P., Hu, Y.: Multi-modal cross-domain alignment network for video moment retrieval. IEEE Trans. Multimedia 25, 7517\u20137532 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"17_CR30","doi-asserted-by":"crossref","unstructured":"Fang, X., Liu, D., Zhou, P., Nan, G.: You can ground earlier than see: an effective and efficient pipeline for temporal sentence grounding in compressed videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2448\u20132460 (2023)","DOI":"10.1109\/CVPR52729.2023.00242"},{"key":"17_CR31","doi-asserted-by":"crossref","unstructured":"Fang, X., Liu, D., Zhou, P., Xu, Z., Li, R.: Hierarchical local-global transformer for temporal sentence grounding. IEEE Trans. Multimedia 26 (2023)","DOI":"10.1109\/TMM.2023.3309551"},{"key":"17_CR32","doi-asserted-by":"crossref","unstructured":"Gao, J., Sun, C., Yang, Z., Nevatia, R.: TALL: temporal activity localization via language query. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5267\u20135275 (2017)","DOI":"10.1109\/ICCV.2017.563"},{"key":"17_CR33","doi-asserted-by":"crossref","unstructured":"Gao, M., Davis, L., Socher, R., Xiong, C.: WSLLN: weakly supervised natural language localization networks. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing, pp. 1481\u20131487 (2019)","DOI":"10.18653\/v1\/D19-1157"},{"key":"17_CR34","doi-asserted-by":"publisher","first-page":"547","DOI":"10.1007\/s001820050125","volume":"28","author":"M Grabisch","year":"1999","unstructured":"Grabisch, M., Roubens, M.: An axiomatic approach to the concept of interaction among players in cooperative games. Int. J. Game Theory 28, 547\u2013565 (1999)","journal-title":"Int. J. Game Theory"},{"key":"17_CR35","doi-asserted-by":"crossref","unstructured":"Guo, C., Liu, D., Zhou, P.: A hybird alignment loss for temporal moment localization with natural language. In: 2022 IEEE International Conference on Multimedia and Expo (ICME), pp.\u00a01\u20136. IEEE (2022)","DOI":"10.1109\/ICME52920.2022.9859675"},{"issue":"7","key":"17_CR36","doi-asserted-by":"publisher","first-page":"6238","DOI":"10.1109\/TCSVT.2024.3358415","volume":"34","author":"D Guo","year":"2024","unstructured":"Guo, D., Li, K., Hu, B., Zhang, Y., Wang, M.: Benchmarking micro-action recognition: dataset, method, and application. IEEE Trans. Circuits Syst. Video Technol. 34(7), 6238\u20136252 (2024)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"17_CR37","doi-asserted-by":"crossref","unstructured":"Hendricks, L.A., Wang, O., Shechtman, E., Sivic, J., Darrell, T., Russell, B.: Localizing moments in video with temporal language. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 1380\u20131390 (2018)","DOI":"10.18653\/v1\/D18-1168"},{"key":"17_CR38","doi-asserted-by":"crossref","unstructured":"Huang, J., Liu, Y., Gong, S., Jin, H.: Cross-sentence temporal and semantic relations in video activity localisation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7199\u20137208 (2021)","DOI":"10.1109\/ICCV48922.2021.00711"},{"key":"17_CR39","doi-asserted-by":"crossref","unstructured":"Jiang, L., Wang, C., Ning, X., Yu, Z.: LTTPoint: a MLP-based point cloud classification method with local topology transformation module. In: 2023 7th Asian Conference on Artificial Intelligence Technology (ACAIT), pp. 783\u2013789. IEEE (2023)","DOI":"10.1109\/ACAIT60137.2023.10528609"},{"key":"17_CR40","doi-asserted-by":"crossref","unstructured":"Jin, P., et al.: Video-text as game players: hierarchical Banzhaf interaction for cross-modal representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2472\u20132482 (2023)","DOI":"10.1109\/CVPR52729.2023.00244"},{"key":"17_CR41","doi-asserted-by":"publisher","first-page":"106485","DOI":"10.1016\/j.ijepes.2020.106485","volume":"125","author":"S Jin","year":"2021","unstructured":"Jin, S., Wang, S., Fang, F.: Game theoretical analysis on capacity configuration for microgrid based on multi-agent system. Int. J. Electr. Power Energy Syst. 125, 106485 (2021)","journal-title":"Int. J. Electr. Power Energy Syst."},{"key":"17_CR42","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"17_CR43","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., Carlos\u00a0Niebles, J.: Dense-captioning events in videos. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 706\u2013715 (2017)","DOI":"10.1109\/ICCV.2017.83"},{"key":"17_CR44","unstructured":"Leech, D.: Computation of Power Indices (2002)"},{"key":"17_CR45","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1007\/BF01254541","volume":"17","author":"E Lehrer","year":"1988","unstructured":"Lehrer, E.: An axiomatization of the Banzhaf value. Int. J. Game Theory 17, 89\u201399 (1988)","journal-title":"Int. J. Game Theory"},{"key":"17_CR46","doi-asserted-by":"crossref","unstructured":"Li, H., Cao, M., Cheng, X., Li, Y., Zhu, Z., Zou, Y.: G2L: semantically aligned and uniform video grounding via geodesic and game theory. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12032\u201312042 (2023)","DOI":"10.1109\/ICCV51070.2023.01105"},{"key":"17_CR47","unstructured":"Li, J., et al.: Fine-grained semantically aligned vision-language pre-training. Adv. Neural Inf. Process. Syst. 35, 7290\u20137303 (2022)"},{"key":"17_CR48","doi-asserted-by":"crossref","unstructured":"Lin, K.Q., et al.: UniVTG: towards unified video-language temporal grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2794\u20132804 (2023)","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"17_CR49","doi-asserted-by":"crossref","unstructured":"Lin, Z., Zhao, Z., Zhang, Z., Wang, Q., Liu, H.: Weakly-supervised video moment retrieval via semantic completion network. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 11539\u201311546 (2020)","DOI":"10.1609\/aaai.v34i07.6820"},{"key":"17_CR50","doi-asserted-by":"crossref","unstructured":"Liu, C., Wen, J., Luo, X., Huang, C., Wu, Z., Xu, Y.: DICNet: deep instance-level contrastive network for double incomplete multi-view multi-label classification. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 8807\u20138815 (2023)","DOI":"10.1609\/aaai.v37i7.26059"},{"key":"17_CR51","doi-asserted-by":"crossref","unstructured":"Liu, C., Wen, J., Luo, X., Xu, Y.: Incomplete multi-view multi-label learning via label-guided masked view and category-aware transformers. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 8816\u20138824 (2023)","DOI":"10.1609\/aaai.v37i7.26060"},{"key":"17_CR52","doi-asserted-by":"crossref","unstructured":"Liu, C., Wen, J., Wu, Z., Luo, X., Huang, C., Xu, Y.: Information recovery-driven deep incomplete multiview clustering network. In: IEEE Transactions on Neural Networks and Learning Systems, pp. 1\u201311 (2023)","DOI":"10.1109\/TNNLS.2023.3286918"},{"key":"17_CR53","doi-asserted-by":"publisher","first-page":"8539","DOI":"10.1109\/TMM.2023.3238514","volume":"25","author":"D Liu","year":"2023","unstructured":"Liu, D., Fang, X., Hu, W., Zhou, P.: Exploring optical-flow-guided motion and detection-based appearance for temporal sentence grounding. IEEE Trans. Multimedia 25, 8539\u20138553 (2023)","journal-title":"IEEE Trans. Multimedia"},{"key":"17_CR54","doi-asserted-by":"crossref","unstructured":"Liu, D., et al.: Unsupervised domain adaptative temporal sentence localization with mutual information maximization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 3567\u20133575 (2024)","DOI":"10.1609\/aaai.v38i4.28145"},{"key":"17_CR55","doi-asserted-by":"crossref","unstructured":"Liu, D., Fang, X., Zhou, P., Di, X., Lu, W., Cheng, Y.: Hypotheses tree building for one-shot temporal sentence localization. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 1640\u20131648 (2023)","DOI":"10.1609\/aaai.v37i2.25251"},{"key":"17_CR56","unstructured":"Liu, D., Hu, W.: Learning to focus on the foreground for temporal sentence grounding. In: Proceedings of the 29th International Conference on Computational Linguistics, pp. 5532\u20135541 (2022)"},{"key":"17_CR57","doi-asserted-by":"crossref","unstructured":"Liu, D., Hu, W.: Skimming, locating, then perusing: a human-like framework for natural language video localization. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4536\u20134545 (2022)","DOI":"10.1145\/3503161.3547782"},{"key":"17_CR58","doi-asserted-by":"crossref","unstructured":"Liu, D., et al.: Filling the information gap between video and query for language-driven moment retrieval. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 4190\u20134199 (2023)","DOI":"10.1145\/3581783.3612038"},{"key":"17_CR59","doi-asserted-by":"crossref","unstructured":"Liu, D., et al.: Context-aware biaffine localizing network for temporal sentence grounding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11235\u201311244 (2021)","DOI":"10.1109\/CVPR46437.2021.01108"},{"issue":"4","key":"17_CR60","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3634749","volume":"20","author":"D Liu","year":"2024","unstructured":"Liu, D., et al.: Transform-equivariant consistency learning for temporal sentence grounding. ACM Trans. Multimedia Comput. Commun. Appl. 20(4), 1\u201319 (2024)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"key":"17_CR61","unstructured":"Liu, D., et al.: Towards robust temporal activity localization learning with noisy labels. In: Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024), pp. 16630\u201316642 (2024)"},{"key":"17_CR62","doi-asserted-by":"crossref","unstructured":"Liu, D., Qu, X., Hu, W.: Reducing the vision and language bias for temporal sentence grounding. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4092\u20134101 (2022)","DOI":"10.1145\/3503161.3547969"},{"key":"17_CR63","doi-asserted-by":"crossref","unstructured":"Liu, D., Qu, X., Liu, X.Y., Dong, J., Zhou, P., Xu, Z.: Jointly cross-and self-modal graph attention network for query-based moment localization. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4070\u20134078 (2020)","DOI":"10.1145\/3394171.3414026"},{"key":"17_CR64","doi-asserted-by":"crossref","unstructured":"Liu, D., Zhou, P.: Jointly visual-and semantic-aware graph memory networks for temporal sentence localization in videos. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10096382"},{"issue":"5","key":"17_CR65","doi-asserted-by":"publisher","first-page":"2491","DOI":"10.1109\/TCSVT.2022.3223725","volume":"33","author":"D Liu","year":"2022","unstructured":"Liu, D., Zhou, P., Xu, Z., Wang, H., Li, R.: Few-shot temporal sentence grounding via memory-guided semantic learning. IEEE Trans. Circuits Syst. Video Technol. 33(5), 2491\u20132505 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"17_CR66","doi-asserted-by":"crossref","unstructured":"Liu, D., et al.: Conditional video diffusion network for fine-grained temporal sentence grounding. IEEE Trans. Multimedia 26 (2023)","DOI":"10.1109\/TMM.2023.3334019"},{"key":"17_CR67","unstructured":"Lundberg, S.M., Lee, S.I.: A unified approach to interpreting model predictions. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, pp. 4768\u20134777 (2017)"},{"key":"17_CR68","doi-asserted-by":"crossref","unstructured":"Ma, M., Yoon, S., Kim, J., Lee, Y., Kang, S., Yoo, C.D.: VLANet: video-language alignment network for weakly-supervised video moment retrieval. In: Proceedings of the European Conference on Computer Vision, pp. 156\u2013171 (2020)","DOI":"10.1007\/978-3-030-58604-1_10"},{"key":"17_CR69","doi-asserted-by":"crossref","unstructured":"Ma, W.C., Huang, D.A., Lee, N., Kitani, K.M.: Forecasting interactive dynamics of pedestrians with fictitious play. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 774\u2013782 (2017)","DOI":"10.1109\/CVPR.2017.493"},{"key":"17_CR70","doi-asserted-by":"crossref","unstructured":"Ma, Y., Liu, Y., Wang, L., Kang, W., Qiao, Y., Wang, Y.: Dual masked modeling for weakly-supervised temporal boundary discovery. IEEE Trans. Multimedia 26 (2023)","DOI":"10.1109\/TMM.2023.3338084"},{"issue":"1\u20132","key":"17_CR71","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1016\/S0304-3975(00)00251-6","volume":"263","author":"Y Matsui","year":"2001","unstructured":"Matsui, Y., Matsui, T.: NP-completeness for calculating power indices of weighted majority games. Theoret. Comput. Sci. 263(1\u20132), 305\u2013310 (2001)","journal-title":"Theoret. Comput. Sci."},{"key":"17_CR72","doi-asserted-by":"publisher","first-page":"607","DOI":"10.1613\/jair.3806","volume":"46","author":"TP Michalak","year":"2013","unstructured":"Michalak, T.P., Aadithya, K.V., Szczepanski, P.L., Ravindran, B., Jennings, N.R.: Efficient computation of the Shapley value for game-theoretic network centrality. J. Artif. Intell. Res. 46, 607\u2013650 (2013)","journal-title":"J. Artif. Intell. Res."},{"key":"17_CR73","doi-asserted-by":"crossref","unstructured":"Mithun, N.C., Paul, S., Roy-Chowdhury, A.K.: Weakly supervised video moment retrieval from text queries. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11592\u201311601 (2019)","DOI":"10.1109\/CVPR.2019.01186"},{"key":"17_CR74","doi-asserted-by":"crossref","unstructured":"Mun, J., Cho, M., Han, B.: Local-global video-text interactions for temporal grounding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 10810\u201310819 (2020)","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"17_CR75","doi-asserted-by":"crossref","unstructured":"Ning, E., Wang, C., Zhang, H., Ning, X., Tiwari, P.: Occluded person re-identification with deep learning: a survey and perspectives. Exp. Syst. Appl. 239, 122419 (2023)","DOI":"10.1016\/j.eswa.2023.122419"},{"key":"17_CR76","doi-asserted-by":"publisher","first-page":"532","DOI":"10.1016\/j.neunet.2023.11.003","volume":"169","author":"E Ning","year":"2024","unstructured":"Ning, E., Wang, Y., Wang, C., Zhang, H., Ning, X.: Enhancement, integration, expansion: activating representation of detailed features for occluded person re-identification. Neural Netw. 169, 532\u2013541 (2024)","journal-title":"Neural Netw."},{"key":"17_CR77","doi-asserted-by":"publisher","first-page":"102467","DOI":"10.1016\/j.displa.2023.102467","volume":"79","author":"E Ning","year":"2023","unstructured":"Ning, E., Zhang, C., Wang, C., Ning, X., Chen, H., Bai, X.: Pedestrian re-ID based on feature consistency and contrast enhancement. Displays 79, 102467 (2023)","journal-title":"Displays"},{"key":"17_CR78","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/BF01262517","volume":"26","author":"AS Nowak","year":"1997","unstructured":"Nowak, A.S.: On an axiomatization of the Banzhaf value without the additivity axiom. Int. J. Game Theory 26, 137\u2013141 (1997)","journal-title":"Int. J. Game Theory"},{"key":"17_CR79","unstructured":"Oord, A.V.D., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"17_CR80","unstructured":"Osborne, M.J., Rubinstein, A.: A Course in Game Theory. MIT Press (1994)"},{"key":"17_CR81","doi-asserted-by":"crossref","unstructured":"Patel, R., Garnelo, M., Gemp, I., Dyer, C., Bachrach, Y.: Game-theoretic vocabulary selection via the Shapley value and Banzhaf index. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 2789\u20132798 (2021)","DOI":"10.18653\/v1\/2021.naacl-main.223"},{"key":"17_CR82","unstructured":"Pavan, M., Pelillo, M.: A new graph-theoretic approach to clustering and segmentation. In: 2003 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, 2003, Proceedings, vol.\u00a01, p.\u00a0I. IEEE (2003)"},{"key":"17_CR83","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: GloVe: global vectors for word representation. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"17_CR84","doi-asserted-by":"crossref","unstructured":"Rodola, E., Bronstein, A.M., Albarelli, A., Bergamasco, F., Torsello, A.: A game-theoretic approach to deformable shape matching. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition, pp. 182\u2013189. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6247674"},{"key":"17_CR85","doi-asserted-by":"crossref","unstructured":"Shapley, L.S., et\u00a0al.: A Value for n-Person Games (1953)","DOI":"10.1515\/9781400881970-018"},{"key":"17_CR86","doi-asserted-by":"crossref","unstructured":"Sigurdsson, G.A., Varol, G., Wang, X., Farhadi, A., Laptev, I., Gupta, A.: Hollywood in homes: crowdsourcing data collection for activity understanding. In: European Conference on Computer Vision, pp. 510\u2013526 (2016)","DOI":"10.1007\/978-3-319-46448-0_31"},{"key":"17_CR87","doi-asserted-by":"publisher","first-page":"126625","DOI":"10.1016\/j.neucom.2023.126625","volume":"554","author":"Y Song","year":"2023","unstructured":"Song, Y., et al.: MARN: multi-level attentional reconstruction networks for weakly supervised video temporal grounding. Neurocomputing 554, 126625 (2023)","journal-title":"Neurocomputing"},{"key":"17_CR88","unstructured":"Song, Y., Wang, J., Ma, L., Yu, Z., Yu, J.: Weakly-supervised multi-level attentional reconstruction network for grounding textual queries in videos. arXiv preprint arXiv:2003.07048 (2020)"},{"key":"17_CR89","doi-asserted-by":"crossref","unstructured":"Tan, R., Xu, H., Saenko, K., Plummer, B.A.: LoGAN: latent graph co-attention network for weakly-supervised video moment retrieval. In: Proceedings of the IEEE Winter Conference on Applications of Computer Vision, pp. 2083\u20132092 (2021)","DOI":"10.1109\/WACV48630.2021.00213"},{"issue":"11","key":"17_CR90","doi-asserted-by":"publisher","first-page":"5577","DOI":"10.1007\/s00371-022-02682-0","volume":"39","author":"K Tang","year":"2023","unstructured":"Tang, K., et al.: RepPVConv: attentively fusing reparameterized voxel features for efficient 3D point cloud perception. Vis. Comput. 39(11), 5577\u20135588 (2023)","journal-title":"Vis. Comput."},{"key":"17_CR91","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2024.3378652","author":"K Tang","year":"2024","unstructured":"Tang, K., Lou, T., Peng, W., Chen, N., Shi, Y., Wang, W.: Effective single-step adversarial training with energy-based models. IEEE Trans. Emerg. Top. Comput. Intell. (2024). https:\/\/doi.org\/10.1109\/TETCI.2024.3378652","journal-title":"IEEE Trans. Emerg. Top. Comput. Intell."},{"key":"17_CR92","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3196129","author":"K Tang","year":"2022","unstructured":"Tang, K., et al.: Decision fusion networks for image classification. IEEE Trans. Neural Netw. Learn. Syst. (2022). https:\/\/doi.org\/10.1109\/TNNLS.2022.3196129","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"6","key":"17_CR93","doi-asserted-by":"publisher","first-page":"5158","DOI":"10.1109\/JIOT.2022.3222159","volume":"10","author":"K Tang","year":"2022","unstructured":"Tang, K., et al.: Rethinking perturbation directions for imperceptible adversarial attacks on point clouds. IEEE Internet Things J. 10(6), 5158\u20135169 (2022)","journal-title":"IEEE Internet Things J."},{"key":"17_CR94","doi-asserted-by":"publisher","unstructured":"Tang, K., et al.: Reparameterization head for efficient multi-input networks. In: ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6190\u20136194 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10447574","DOI":"10.1109\/ICASSP48485.2024.10447574"},{"key":"17_CR95","doi-asserted-by":"crossref","unstructured":"Tang, K., et al.: Reparameterization head for efficient multi-input networks. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6190\u20136194. IEEE (2024)","DOI":"10.1109\/ICASSP48485.2024.10447574"},{"key":"17_CR96","doi-asserted-by":"crossref","unstructured":"Torsello, A., Bulo, S.R., Pelillo, M.: Grouping with asymmetric affinities: a game-theoretic perspective. In: 2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201906), vol.\u00a01, pp. 292\u2013299. IEEE (2006)","DOI":"10.1109\/CVPR.2006.130"},{"key":"17_CR97","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3D convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4489\u20134497 (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"17_CR98","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"key":"17_CR99","doi-asserted-by":"crossref","unstructured":"Wang, C., Ning, X., Li, W., Bai, X., Gao, X.: 3D person re-identification based on global semantic guidance and local feature aggregation. IEEE Trans. Circuits Syst. Video Technol. 34(6) (2023)","DOI":"10.1109\/TCSVT.2023.3328712"},{"key":"17_CR100","first-page":"1","volume":"60","author":"C Wang","year":"2022","unstructured":"Wang, C., Ning, X., Sun, L., Zhang, L., Li, W., Bai, X.: Learning discriminative features by covering local geometric space for point cloud analysis. IEEE Trans. Geosci. Remote Sens. 60, 1\u201315 (2022)","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"17_CR101","doi-asserted-by":"publisher","first-page":"102080","DOI":"10.1016\/j.displa.2021.102080","volume":"70","author":"C Wang","year":"2021","unstructured":"Wang, C., Wang, C., Li, W., Wang, H.: A brief survey on RGB-D semantic segmentation using deep learning. Displays 70, 102080 (2021)","journal-title":"Displays"},{"issue":"4","key":"17_CR102","first-page":"1962","volume":"34","author":"C Wang","year":"2022","unstructured":"Wang, C., Wang, H., Ning, X., Shengwei, T., Li, W.: 3D point cloud classification method based on dynamic coverage of local area. J. Softw. 34(4), 1962\u20131976 (2022)","journal-title":"J. Softw."},{"key":"17_CR103","doi-asserted-by":"crossref","unstructured":"Wang, J., Ma, L., Jiang, W.: Temporally grounding language queries in videos by contextual boundary-aware prediction. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 12168\u201312175 (2020)","DOI":"10.1609\/aaai.v34i07.6897"},{"key":"17_CR104","doi-asserted-by":"publisher","first-page":"3276","DOI":"10.1109\/TMM.2021.3096087","volume":"24","author":"Y Wang","year":"2021","unstructured":"Wang, Y., Deng, J., Zhou, W., Li, H.: Weakly supervised temporal adjacent network for language grounding. IEEE Trans. Multimedia 24, 3276\u20133286 (2021)","journal-title":"IEEE Trans. Multimedia"},{"key":"17_CR105","doi-asserted-by":"crossref","unstructured":"Wang, Z., Chen, J., Jiang, Y.G.: Visual co-occurrence alignment learning for weakly-supervised video moment retrieval. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 1459\u20131468 (2021)","DOI":"10.1145\/3474085.3475278"},{"key":"17_CR106","doi-asserted-by":"publisher","unstructured":"Wen, J., et al.: Deep double incomplete multi-view multi-label learning with incomplete labels and missing views. IEEE Trans. Neural Netw. Learn. Syst. 35(8), 1\u201313 (2023). https:\/\/doi.org\/10.1109\/TNNLS.2023.3260349","DOI":"10.1109\/TNNLS.2023.3260349"},{"key":"17_CR107","doi-asserted-by":"crossref","unstructured":"Wen, J., Zhang, Z., Li, Z.J.: A survey on incomplete multiview clustering. IEEE Trans. Syst. Man Cybern. Syst. 53(2), 1136\u20131149 (2023)","DOI":"10.1109\/TSMC.2022.3192635"},{"key":"17_CR108","doi-asserted-by":"crossref","unstructured":"Winter, E.: The shapley value. In: Handbook of Game Theory with Economic Applications, vol. 3, pp. 2025\u20132054 (2002)","DOI":"10.1016\/S1574-0005(02)03016-3"},{"key":"17_CR109","doi-asserted-by":"crossref","unstructured":"Wu, H., et al.: Atomic-action-based contrastive network for weakly supervised temporal language grounding. In: 2023 IEEE International Conference on Multimedia and Expo (ICME), pp. 1523\u20131528. IEEE (2023)","DOI":"10.1109\/ICME55011.2023.00263"},{"key":"17_CR110","doi-asserted-by":"crossref","unstructured":"Xiong, Z., Liu, D., Zhou, P.: Gaussian kernel-based cross modal network for spatio-temporal video grounding. In: IEEE International Conference on Image Processing (ICIP), pp. 2481\u20132485 (2022)","DOI":"10.1109\/ICIP46576.2022.9897707"},{"key":"17_CR111","doi-asserted-by":"crossref","unstructured":"Xiong, Z., Liu, D., Zhou, P., Zhu, J.: Tracking objects and activities with attention for temporal sentence grounding. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10096206"},{"key":"17_CR112","doi-asserted-by":"publisher","first-page":"3252","DOI":"10.1109\/TIP.2021.3058614","volume":"30","author":"W Yang","year":"2021","unstructured":"Yang, W., Zhang, T., Zhang, Y., Wu, F.: Local correspondence network for weakly supervised temporal sentence grounding. IEEE Trans. Image Process. 30, 3252\u20133262 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"17_CR113","doi-asserted-by":"crossref","unstructured":"Yu, Z., Li, L., Xie, J., Wang, C., Li, W., Ning, X.: Pedestrian 3D shape understanding for person re-identification via multi-view learning. IEEE Trans. Circuits Syst. Video Technol. 34(7) (2024)","DOI":"10.1109\/TCSVT.2024.3358850"},{"key":"17_CR114","unstructured":"Yuan, Y., Ma, L., Wang, J., Liu, W., Zhu, W.: Semantic conditioned dynamic modulation for temporal sentence grounding in videos. In: Proceedings of the 33rd International Conference on Neural Information Processing Systems, pp. 536\u2013546 (2019)"},{"key":"17_CR115","doi-asserted-by":"crossref","unstructured":"Zeng, R., Xu, H., Huang, W., Chen, P., Tan, M., Gan, C.: Dense regression network for video grounding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 10287\u201310296 (2020)","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"17_CR116","doi-asserted-by":"crossref","unstructured":"Zhang, D., Dai, X., Wang, X., Wang, Y.F., Davis, L.S.: MAN: moment alignment network for natural language moment retrieval via iterative graph adjustment. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1247\u20131257 (2019)","DOI":"10.1109\/CVPR.2019.00134"},{"key":"17_CR117","doi-asserted-by":"crossref","unstructured":"Zhang, H., Sun, A., Jing, W., Zhou, J.T.: Span-based localizing network for natural language video localization. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 6543\u20136554 (2020)","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"17_CR118","doi-asserted-by":"crossref","unstructured":"Zhang, H., Xie, Y., Zheng, L., Zhang, D., Zhang, Q.: Interpreting multivariate Shapley interactions in DNNs. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 10877\u201310886 (2021)","DOI":"10.1609\/aaai.v35i12.17299"},{"key":"17_CR119","doi-asserted-by":"publisher","first-page":"102456","DOI":"10.1016\/j.displa.2023.102456","volume":"79","author":"H Zhang","year":"2023","unstructured":"Zhang, H., et al.: Deep learning-based 3D point cloud classification: a systematic survey and outlook. Displays 79, 102456 (2023)","journal-title":"Displays"},{"key":"17_CR120","doi-asserted-by":"crossref","unstructured":"Zhang, H., Wang, C., Yu, L., Tian, S., Ning, X., Rodrigues, J.: PointGT: a method for point-cloud classification and segmentation based on local geometric transformation. IEEE Trans. Multimedia 26 (2024)","DOI":"10.1109\/TMM.2024.3374580"},{"key":"17_CR121","doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2D temporal adjacent networks for moment localization with natural language. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 12870\u201312877 (2020)","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"17_CR122","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Lin, Z., Zhao, Z., Xiao, Z.: Cross-modal interaction networks for query-based moment retrieval in videos. In: Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 655\u2013664 (2019)","DOI":"10.1145\/3331184.3331235"},{"key":"17_CR123","first-page":"18123","volume":"33","author":"Z Zhang","year":"2020","unstructured":"Zhang, Z., Zhao, Z., Lin, Z., He, X., et al.: Counterfactual contrastive learning for weakly-supervised vision-language grounding. Adv. Neural. Inf. Process. Syst. 33, 18123\u201318134 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"17_CR124","doi-asserted-by":"crossref","unstructured":"Zheng, M., Huang, Y., Chen, Q., Liu, Y.: Weakly supervised video moment localization with contrastive negative sample mining. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 3517\u20133525 (2022)","DOI":"10.1609\/aaai.v36i3.20263"},{"key":"17_CR125","doi-asserted-by":"crossref","unstructured":"Zheng, M., Huang, Y., Chen, Q., Peng, Y., Liu, Y.: Weakly supervised temporal sentence grounding with Gaussian-based contrastive proposal learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15555\u201315564 (2022)","DOI":"10.1109\/CVPR52688.2022.01511"},{"issue":"2","key":"17_CR126","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3544493","volume":"19","author":"Q Zheng","year":"2023","unstructured":"Zheng, Q., et al.: Progressive localization networks for language-based moment localization. ACM Trans. Multimedia Comput. Commun. Appl. 19(2), 1\u201321 (2023)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"key":"17_CR127","doi-asserted-by":"crossref","unstructured":"Zhu, J., et\u00a0al.: Rethinking the video sampling and reasoning strategies for temporal sentence grounding. arXiv preprint arXiv:2301.00514 (2023)","DOI":"10.18653\/v1\/2022.findings-emnlp.41"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72995-9_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T20:05:03Z","timestamp":1732392303000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72995-9_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,24]]},"ISBN":["9783031729942","9783031729959"],"references-count":127,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72995-9_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,24]]},"assertion":[{"value":"24 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}