{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T16:08:31Z","timestamp":1743091711063,"version":"3.40.3"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031784552"},{"type":"electronic","value":"9783031784569"}],"license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-78456-9_20","type":"book-chapter","created":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T11:25:01Z","timestamp":1733138701000},"page":"308-324","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Size-Modulated Deformable Attention in Spatio-Temporal Video Grounding Pipelines"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-5661-7022","authenticated-orcid":false,"given":"Hans","family":"Tiwari","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8658-8890","authenticated-orcid":false,"given":"Selen","family":"Pehlivan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7218-3131","authenticated-orcid":false,"given":"Jorma","family":"Laaksonen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,3]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: ECCV. pp. 213\u2013229. Springer (2020)","key":"20_CR1","DOI":"10.1007\/978-3-030-58452-8_13"},{"doi-asserted-by":"crossref","unstructured":"Chen, J., Chen, X., Ma, L., Jie, Z., Chua, T.S.: Temporally grounding natural sentence in video. In: ENLP. pp. 162\u2013171 (2018)","key":"20_CR2","DOI":"10.18653\/v1\/D18-1015"},{"doi-asserted-by":"crossref","unstructured":"Chen, J., Ma, L., Chen, X., Jie, Z., Luo, J.: Localizing natural language in videos. In: AAAI. vol.\u00a033, pp. 8175\u20138182 (2019)","key":"20_CR3","DOI":"10.1609\/aaai.v33i01.33018175"},{"doi-asserted-by":"crossref","unstructured":"Chen, Z., Ma, L., Luo, W., Wong, K.Y.K.: Weakly-supervised spatio-temporally grounding natural sentence in video. arXiv preprint arXiv:1906.02549 (2019)","key":"20_CR4","DOI":"10.18653\/v1\/P19-1183"},{"unstructured":"Gao, J., et\u00a0al.: Fast, accurate, and lightweight temporal action proposal generation. In: ECCV (2020)","key":"20_CR5"},{"key":"20_CR6","first-page":"22605","volume":"33","author":"S Ging","year":"2020","unstructured":"Ging, S., Zolfaghari, M., Pirsiavash, H., Brox, T.: COOT: Cooperative hierarchical transformer for video-text representation learning. NIPS 33, 22605\u201322618 (2020)","journal-title":"NIPS"},{"doi-asserted-by":"crossref","unstructured":"Gu, X., Fan, H., Huang, Y., Luo, T., Zhang, L.: Context-guided spatio-temporal video grounding. In: CVPR. pp. 18330\u201318339 (2024)","key":"20_CR7","DOI":"10.1109\/CVPR52733.2024.01735"},{"issue":"1","key":"20_CR8","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","volume":"45","author":"K Han","year":"2023","unstructured":"Han, K., Wang, Y., Chen, H., Chen, X., Guo, J., Liu, Z., Tang, Y., Xiao, A., Xu, C., Xu, Y., Yang, Z., Zhang, Y., Tao, D.: A survey on vision transformer. IEEE Trans. Pattern Anal. Mach. Intell. 45(1), 87\u2013110 (2023). https:\/\/doi.org\/10.1109\/TPAMI.2022.3152247","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"20_CR9","first-page":"29192","volume":"35","author":"Y Jin","year":"2022","unstructured":"Jin, Y., Yuan, Z., Mu, Y., et al.: Embracing consistency: A one-stage approach for spatio-temporal video grounding. NIPS 35, 29192\u201329204 (2022)","journal-title":"NIPS"},{"doi-asserted-by":"crossref","unstructured":"Kamath, A., Singh, M., LeCun, Y., Synnaeve, G., Misra, I., Carion, N.: MDETR \u2013 modulated detection for end-to-end multi-modal understanding. In: ICCV. pp. 1780\u20131790 (2021)","key":"20_CR10","DOI":"10.1109\/ICCV48922.2021.00180"},{"doi-asserted-by":"crossref","unstructured":"Liang, Y., Liang, X., Tang, Y., Yang, Z., Li, Z., Wang, J., Ding, W., Huang, S.L.: Costa: End-to-end comprehensive space-time entanglement for spatio-temporal video grounding. In: AAAI. vol.\u00a038, pp. 3324\u20133332 (2024)","key":"20_CR11","DOI":"10.1609\/aaai.v38i4.28118"},{"doi-asserted-by":"crossref","unstructured":"Lin, Z., Tan, C., Hu, J.F., Jin, Z., Ye, T., Zheng, W.S.: Collaborative static and dynamic vision-language streams for spatio-temporal video grounding. In: CVPR. pp. 23100\u201323109 (2023)","key":"20_CR12","DOI":"10.1109\/CVPR52729.2023.02212"},{"unstructured":"Liu, S., Li, F., Zhang, H., Yang, X., Qi, X., Su, H., Zhu, J., Zhang, L.: DAB-DETR: Dynamic anchor boxes are better queries for DETR. arXiv preprint arXiv:2201.12329 (2022)","key":"20_CR13"},{"unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., Stoyanov, V.: RoBERTa: A robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)","key":"20_CR14"},{"unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (2019)","key":"20_CR15"},{"unstructured":"Munasinghe, S., Thushara, R., Maaz, M., Rasheed, H.A., Khan, S., Shah, M., Khan, F.: Pg-video-llava: Pixel grounding large video-language models. arXiv preprint arXiv:2311.13435 (2023)","key":"20_CR16"},{"unstructured":"Nag, S., Zhu, X., Xiang, T.: Few-shot temporal action localization with query adaptive transformer. arXiv preprint arXiv:2110.10552 (2021)","key":"20_CR17"},{"doi-asserted-by":"crossref","unstructured":"Shang, X., Di, D., Xiao, J., Cao, Y., Yang, X., Chua, T.S.: Annotating objects and relations in user-generated videos. In: Proceedings of the 2019 on International Conference on Multimedia Retrieval. pp. 279\u2013287. ACM (2019)","key":"20_CR18","DOI":"10.1145\/3323873.3325056"},{"doi-asserted-by":"crossref","unstructured":"Su, R., Yu, Q., Xu, D.: STVGBert: A visual-linguistic transformer based framework for spatio-temporal video grounding. In: ICCV. pp. 1513\u20131522 (2021). 10.1109\/ICCV48922.2021.00156","key":"20_CR19","DOI":"10.1109\/ICCV48922.2021.00156"},{"issue":"12","key":"20_CR20","doi-asserted-by":"publisher","first-page":"8238","DOI":"10.1109\/TCSVT.2021.3085907","volume":"32","author":"Z Tang","year":"2021","unstructured":"Tang, Z., Liao, Y., Liu, S., Li, G., Jin, X., Jiang, H., Yu, Q., Xu, D.: Human-centric spatio-temporal video grounding with visual transformers. IEEE Trans. Circuits Syst. Video Technol. 32(12), 8238\u20138249 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"2","key":"20_CR21","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1145\/2812802","volume":"59","author":"B Thomee","year":"2016","unstructured":"Thomee, B., Shamma, D.A., Friedland, G., Elizalde, B., Ni, K., Poland, D., Borth, D., Li, L.J.: Yfcc100m: The new data in multimedia research. Commun. ACM 59(2), 64\u201373 (2016)","journal-title":"Commun. ACM"},{"unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: NIPS (2017)","key":"20_CR22"},{"doi-asserted-by":"crossref","unstructured":"Yang, A., Miech, A., Sivic, J., Laptev, I., Schmid, C.: TubeDETR: Spatio-temporal video grounding with transformers. In: CVPR. pp. 16442\u201316453 (2022)","key":"20_CR23","DOI":"10.1109\/CVPR52688.2022.01595"},{"doi-asserted-by":"crossref","unstructured":"Zhang, S., Peng, H., Fu, J., Luo, J.: Learning 2D temporal adjacent networks for moment localization with natural language. In: AAAI. vol.\u00a034, pp. 12870\u201312877 (2020)","key":"20_CR24","DOI":"10.1609\/aaai.v34i07.6984"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Z., et\u00a0al.: MAN: Moment alignment network for natural language moment retrieval via iterative graph adjustment. In: CVPR. pp. 1247\u20131255 (2019)","key":"20_CR25","DOI":"10.1109\/CVPR.2019.00134"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Z., et\u00a0al.: Cross-modal interaction networks for query-based moment retrieval in videos. In: ACM SIGIR. pp. 655\u2013664 (2020)","key":"20_CR26","DOI":"10.1145\/3331184.3331235"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhao, Z., Zhao, Y., Wang, Q., Liu, H., Gao, L.: Where does it exist: Spatio-temporal video grounding for multi-form sentences. In: CVPR (2020)","key":"20_CR27","DOI":"10.1109\/CVPR42600.2020.01068"},{"unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: Deformable transformers for End-to-End object detection. arXiv preprint arXiv:2010.04159 (2020)","key":"20_CR28"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-78456-9_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T12:13:01Z","timestamp":1733141581000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-78456-9_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"ISBN":["9783031784552","9783031784569"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-78456-9_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"3 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}