{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T16:30:16Z","timestamp":1785601816205,"version":"3.56.0"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002367","name":"Chinese Academy of Sciences","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002367","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,4]]},"DOI":"10.1109\/icassp49357.2023.10095026","type":"proceedings-article","created":{"date-parts":[[2023,5,5]],"date-time":"2023-05-05T17:28:30Z","timestamp":1683307710000},"page":"1-5","source":"Crossref","is-referenced-by-count":17,"title":["Visual Answer Localization with Cross-Modal Mutual Knowledge Transfer"],"prefix":"10.1109","author":[{"given":"Yixuan","family":"Weng","sequence":"first","affiliation":[{"name":"Chinese Academy Sciences,National Laboratory of Pattern Recognition Institute of Automation"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Li","sequence":"additional","affiliation":[{"name":"Hunan University,College of Electrical and Information Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Debertav3: Improving deberta using electra-style pre-training with gradient-disentangled embedding sharing","author":"he","year":"2021"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093328"},{"key":"ref14","article-title":"Qanet: Combining local convolution with global self-attention for reading comprehension","author":"yu","year":"2018"},{"key":"ref20","article-title":"Decoupled weight decay regularization","author":"loshchilov","year":"2017","journal-title":"Learning"},{"key":"ref11","article-title":"Towards visual-prompt temporal answering grounding in medical instructional video","author":"li","year":"0"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3063631"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-021-0247-3"},{"key":"ref1","first-page":"8","article-title":"Knowledge transfer with visual prompt in multi-modal dialogue understanding and generation","author":"zhu","year":"2022","journal-title":"Proceedings of the First Workshop On Transcript Understanding"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1736"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.324"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210003"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.bionlp-1.43"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.bionlp-1.25"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1108\/INTR-06-2018-0270"},{"key":"ref3","article-title":"Wiki-how: A large scale text summarization dataset","author":"koupaee","year":"2018"},{"key":"ref6","article-title":"A dataset for medical instructional video classification and question answering","author":"gupta","year":"2022"},{"key":"ref5","first-page":"5450","article-title":"Tutorialvqa: Question answering dataset for tutorial videos","author":"colas","year":"2020","journal-title":"Proceedings of the Twelfth Language Re-sources and Evaluation Conference"}],"event":{"name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Rhodes Island, Greece","start":{"date-parts":[[2023,6,4]]},"end":{"date-parts":[[2023,6,10]]}},"container-title":["ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10094559\/10094560\/10095026.pdf?arnumber=10095026","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T18:57:07Z","timestamp":1699901827000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10095026\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,4]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/icassp49357.2023.10095026","relation":{},"subject":[],"published":{"date-parts":[[2023,6,4]]}}}