{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T16:36:54Z","timestamp":1780418214240,"version":"3.54.1"},"reference-count":91,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62221002"],"award-info":[{"award-number":["62221002"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62171183"],"award-info":[{"award-number":["62171183"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004735","name":"Natural Science Foundation of Hunan Province","doi-asserted-by":"publisher","award":["2022JJ20017"],"award-info":[{"award-number":["2022JJ20017"]}],"id":[{"id":"10.13039\/501100004735","id-type":"DOI","asserted-by":"publisher"}]},{"name":"CAAI-Huawei MindSpore Open Fund"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tpami.2024.3411045","type":"journal-article","created":{"date-parts":[[2024,6,7]],"date-time":"2024-06-07T13:27:44Z","timestamp":1717766864000},"page":"8836-8853","source":"Crossref","is-referenced-by-count":30,"title":["Towards Visual-Prompt Temporal Answer Grounding in Instructional Video"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0585-9848","authenticated-orcid":false,"given":"Shutao","family":"Li","sequence":"first","affiliation":[{"name":"College of Electrical and Information Engineering, Hunan University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6508-5071","authenticated-orcid":false,"given":"Bin","family":"Li","sequence":"additional","affiliation":[{"name":"College of Electrical and Information Engineering, Hunan University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7029-8784","authenticated-orcid":false,"given":"Bin","family":"Sun","sequence":"additional","affiliation":[{"name":"College of Electrical and Information Engineering, Hunan University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-8689","authenticated-orcid":false,"given":"Yixuan","family":"Weng","sequence":"additional","affiliation":[{"name":"National Laboratory of Pattern Recognition Institute of Automation, Chinese Academy Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01463"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3505244"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00365"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2980824"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00020"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-023-02036-y"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/s11423-020-09749-6"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.bionlp-1.25"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1167"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1015"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58589-1_27"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1641"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref15","article-title":"DeBERTaV3: Improving deberta using electra-style pre-training with gradient-disentangled embedding sharing","author":"He","year":"2021"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3532626"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3258628"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00207"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"ref22","article-title":"Text-to-clip video retrieval with early fusion and re-captioning","author":"Xu","year":"2018"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018199"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3556537"},{"key":"ref25","first-page":"1984","article-title":"ExCL: Extractive clip localization using natural language descriptions","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics: Hum. Lang. Technol.","author":"Ghosh"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093328"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3063631"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3060449"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/tcsvt.2023.3250518"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3086591"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3478025"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2022.01.001"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00272"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.193"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00638"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00217"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.370"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3173208"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"issue":"8","key":"ref46","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3411763.3451760"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"ref49","first-page":"16158","article-title":"Why do pretrained language models help in downstream tasks? An analysis of head and prompt tuning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wei"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.167"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.346"},{"key":"ref52","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.576"},{"key":"ref54","article-title":"SimVLM: Simple visual language model pretraining with weak supervision","author":"Wang","year":"2021"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2024.01.004"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-26316-3_33"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414567"},{"key":"ref61","article-title":"Neural machine translation by jointly learning to align and translate","author":"Bahdanau","year":"2014"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1264"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1423"},{"key":"ref64","first-page":"5450","article-title":"TutorialVQA: Question answering dataset for tutorial videos","volume-title":"Proc. 12th Lang. Resour. Eval. Conf.","author":"Colas"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1736"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210003"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1158"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.324"},{"key":"ref70","article-title":"Semi-supervised classification with graph convolutional networks","author":"Kipf","year":"2016"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1123"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3120745"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3211850"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3191841"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3532083"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00225"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16406"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547782"},{"key":"ref79","article-title":"RoBERTa: A robustly optimized BERT pretraining approach","author":"Liu","year":"2019"},{"key":"ref80","first-page":"4271","article-title":"Funnel-transformer: Filtering out sequential redundancy for efficient language processing","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dai"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.bionlp-1.43"},{"key":"ref83","first-page":"124","article-title":"Zero-shot video question answering via frozen bidirectional language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref84","article-title":"Deberta: Decoding-enhanced bert with disentangled attention","author":"He","year":"2020"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"ref86","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref88","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"Srivastava","year":"2014","journal-title":"J. Mach. Learn. Res."},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/d14-1181"},{"key":"ref90","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proc. 13th Int. Conf. Artif. Intell. Statist.","author":"Glorot"},{"key":"ref91","article-title":"Efficient estimation of word representations in vector space","author":"Mikolov","year":"2013"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/10746266\/10552074.pdf?arnumber=10552074","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T19:25:39Z","timestamp":1732649139000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10552074\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":91,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3411045","relation":{"has-preprint":[{"id-type":"doi","id":"10.36227\/techrxiv.22182736.v1","asserted-by":"object"}]},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}