{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T06:45:59Z","timestamp":1785653159012,"version":"3.56.0"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_16","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:47:05Z","timestamp":1785649625000},"page":"235-250","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["ClipTBP: Clip-Pair Based Temporal Boundary Prediction with Boundary-Aware Learning for Moment Retrieval"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-7319-4664","authenticated-orcid":false,"given":"Ji-Hyeon","family":"Kim","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4200-5136","authenticated-orcid":false,"given":"Ho-Joong","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6249-4996","authenticated-orcid":false,"given":"Seong-Whan","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"16_CR1","doi-asserted-by":"crossref","unstructured":"Cao, Z., et al.: FlashVTG: feature layering and adaptive score handling network for video temporal grounding. arXiv preprint arXiv:2412.13441 (2024)","DOI":"10.1109\/WACV61041.2025.00894"},{"key":"16_CR2","doi-asserted-by":"crossref","unstructured":"Carion, N., et al.: End-to-end object detection with transformers. In: Proceedings of the European conference on computer vision (ECCV), pp. 213\u2013229 (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Chen, G., et\u00a0al.: Video mamba suite: state space model as a versatile alternative for video understanding. arXiv preprint arXiv:2403.09626 (2024)","DOI":"10.1007\/s11263-025-02597-y"},{"key":"16_CR4","unstructured":"Escorcia, V., Soldan, M., Sivic, J., Ghanem, B., Russell, B.: Finding moments in video collections using natural language. arXiv preprint arXiv:1907.12763 (2019)"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"16_CR6","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"16_CR7","unstructured":"Gu, A., Goel, K., Re, C.: Efficiently modeling long sequences with structured state spaces. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Hadsell, R., Chopra, S., LeCun, Y.: Dimensionality reduction by learning an invariant mapping. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), vol.\u00a02, pp. 1735\u20131742 (2006)","DOI":"10.1109\/CVPR.2006.100"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Jeong, J.H., et\u00a0al.: Multimodal signal dataset for 11 intuitive movement tasks from single upper extremity during multiple recording sessions. GigaScience 9(10), giaa098 (2020)","DOI":"10.1093\/gigascience\/giaa098"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Jiang, Y., et\u00a0al.: Prior knowledge integration via LLM encoding and pseudo event regulation for video moment retrieval. In: Proceedings of the ACM International Conference on Multimedia (ACM MM), pp. 7249\u20137258 (2024)","DOI":"10.1145\/3664647.3681115"},{"key":"16_CR11","doi-asserted-by":"crossref","unstructured":"Kim, J.H., Lee, S.H., Lee, J.H., Lee, S.W.: Fre-GAN: adversarial frequency-consistent audio synthesis. arXiv preprint arXiv:2106.02297 (2021)","DOI":"10.21437\/Interspeech.2021-845"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Lee, G.H., Lee, S.W.: Uncertainty-aware mesh decoder for high fidelity 3D face reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00614"},{"key":"16_CR13","unstructured":"Lei, J., Berg, T.L., Bansal, M.: Detecting moments and highlights in videos via natural language queries. In: Advances in Neural Information Processing Systems, vol. 34, pp. 11846\u201311858 (2021)"},{"key":"16_CR14","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Berg, T.L., Bansal, M.: TVR: a large-scale dataset for video-subtitle moment retrieval. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 447\u2013463 (2020)","DOI":"10.1007\/978-3-030-58589-1_27"},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Li, P., et al.: MomentDiff: generative video moment retrieval from random to real. In: Advances in Neural Information Processing Systems, vol. 36, pp. 65948\u201365966 (2023)","DOI":"10.52202\/075280-2880"},{"key":"16_CR16","doi-asserted-by":"crossref","unstructured":"Lin, K.Q., et al.: UniVTG: towards unified video-language temporal grounding. In: Proceedings of the IEEE\/CVF International Conference on Learning Representations (ICCV), pp. 2794\u20132804 (2023)","DOI":"10.1109\/ICCV51070.2023.00262"},{"key":"16_CR17","doi-asserted-by":"crossref","unstructured":"Liu, Y., et\u00a0al.: $$R^2$$-tuning: efficient image-to-video transfer learning for video temporal grounding. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 421\u2013438 (2024)","DOI":"10.1007\/978-3-031-72940-9_24"},{"key":"16_CR18","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: UMT: unified multi-modal transformers for joint video moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3042\u20133051 (2022)","DOI":"10.1109\/CVPR52688.2022.00305"},{"key":"16_CR19","doi-asserted-by":"crossref","unstructured":"Luo, D., Huang, J., Gong, S., Jin, H., Liu, Y.: Zero-shot video moment retrieval from frozen vision-language models. In: Proceedings of the IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 5464\u20135473 (2024)","DOI":"10.1109\/WACV57701.2024.00538"},{"key":"16_CR20","doi-asserted-by":"publisher","first-page":"439","DOI":"10.1016\/j.neunet.2022.08.029","volume":"155","author":"K Min","year":"2022","unstructured":"Min, K., Lee, G.H., Lee, S.W.: Attentional feature pyramid network for small object detection. Neural Netw. 155, 439\u2013450 (2022)","journal-title":"Neural Netw."},{"key":"16_CR21","unstructured":"Moon, W., Hyun, S., Lee, S., Heo, J.P.: Correlation-guided query-dependency calibration for video temporal grounding. arXiv preprint arXiv:2311.08835 (2023)"},{"key":"16_CR22","doi-asserted-by":"crossref","unstructured":"Moon, W., Hyun, S., Park, S., Park, D., Heo, J.P.: Query-dependent video representation for moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 23023\u201323033 (2023)","DOI":"10.1109\/CVPR52729.2023.02205"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Mun, J., Cho, M., Han, B.: Local-global video-text interactions for temporal grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10810\u201310819 (2020)","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"16_CR24","doi-asserted-by":"crossref","unstructured":"Nam, H., Ha, J.W., Kim, J.: Dual attention networks for multimodal reasoning and matching. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 299\u2013307 (2017)","DOI":"10.1109\/CVPR.2017.232"},{"key":"16_CR25","doi-asserted-by":"crossref","unstructured":"Pan, Y., Zhang, Y., Zhao, X.: FAWL: weakly-supervised video corpus moment retrieval with frame-wise auxiliary alignment and weighted contrastive learning. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2025)","DOI":"10.1109\/ICASSP49660.2025.10887823"},{"key":"16_CR26","doi-asserted-by":"crossref","unstructured":"Panta, L., et al.: Cross-modal contrastive learning with asymmetric co-attention network for video moment retrieval. In: Proceedings of the IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 607\u2013614 (2024)","DOI":"10.1109\/WACVW60836.2024.00071"},{"key":"16_CR27","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the International Conference on Machine Learning (ICML), vol. 139, pp. 8748\u20138763 (2021)"},{"key":"16_CR28","doi-asserted-by":"crossref","unstructured":"Regneri, M., et al.: Grounding action descriptions in videos. Trans. Assoc. Comput. Linguist. (TACL) 1, 25\u201336 (2013)","DOI":"10.1162\/tacl_a_00207"},{"key":"16_CR29","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: FaceNet: a unified embedding for face recognition and clustering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 815\u2013823 (2015)","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"16_CR30","doi-asserted-by":"publisher","first-page":"106642","DOI":"10.1016\/j.neunet.2024.106642","volume":"180","author":"D Shrewsbury","year":"2024","unstructured":"Shrewsbury, D., Kim, S., Lee, S.W.: Adaptive ambiguity-aware weighting for multi-label recognition with limited annotations. Neural Netw. 180, 106642 (2024)","journal-title":"Neural Netw."},{"key":"16_CR31","doi-asserted-by":"crossref","unstructured":"Sun, H., Zhou, M., Chen, W., Xie, W.: TR-DETR: task-reciprocal transformer for joint moment retrieval and highlight detection. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), vol.\u00a038, pp. 4998\u20135007 (2024)","DOI":"10.1609\/aaai.v38i5.28304"},{"key":"16_CR32","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: InternVideo2: scaling foundation models for multimodal video understanding. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 396\u2013416 (2024)","DOI":"10.1007\/978-3-031-73013-9_23"},{"key":"16_CR33","doi-asserted-by":"crossref","unstructured":"Xiao, Y., et al.: Bridging the Gap: a unified video comprehension framework for moment retrieval and highlight detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18709\u201318719 (2024)","DOI":"10.1109\/CVPR52733.2024.01770"},{"key":"16_CR34","unstructured":"Xu, H., He, K., Sigal, L., Sclaroff, S., Saenko, K.: Text-to-clip video retrieval with early fusion and re-captioning. arXiv preprint arXiv:1804.051132(6), 7 (2018)"},{"key":"16_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, H., et al.: Video corpus moment retrieval with contrastive learning. arXiv preprint arXiv:2105.06247 (2021)","DOI":"10.1145\/3404835.3462874"},{"key":"16_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2D temporal adjacent networks for moment localization with natural language. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), vol.\u00a034, pp. 12870\u201312877 (2020)","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"16_CR37","doi-asserted-by":"crossref","unstructured":"Zhou, X., Wei, F., Duan, L., Yao, A., Li, W.: The devil is in the spurious correlations: boosting moment retrieval with dynamic learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 20981\u201320990 (2025)","DOI":"10.1109\/ICCV51701.2025.01950"},{"key":"16_CR38","unstructured":"Zhu, L., et al.: Vision Mamba: efficient visual representation learning with bidirectional state space model. In: Proceedings of the International Conference on Machine Learning (ICML) (2024)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:47:08Z","timestamp":1785649628000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}