{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T18:52:18Z","timestamp":1763491938640,"version":"3.45.0"},"reference-count":50,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100014219","name":"National Science Fund for Distinguished Young Scholars","doi-asserted-by":"publisher","award":["62025205"],"award-info":[{"award-number":["62025205"]}],"id":[{"id":"10.13039\/501100014219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62032020","61960206008"],"award-info":[{"award-number":["62032020","61960206008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Human-Mach. Syst."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1109\/thms.2025.3588768","type":"journal-article","created":{"date-parts":[[2025,8,5]],"date-time":"2025-08-05T18:07:35Z","timestamp":1754417255000},"page":"726-735","source":"Crossref","is-referenced-by-count":0,"title":["Cinematographic-Aware Coherent Shot Assembly for How-To Vlog Generation"],"prefix":"10.1109","volume":"55","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8173-3844","authenticated-orcid":false,"given":"Yuqi","family":"Zhang","sequence":"first","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6097-2467","authenticated-orcid":false,"given":"Bin","family":"Guo","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6411-4486","authenticated-orcid":false,"given":"Ying","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nuo","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1682-910X","authenticated-orcid":false,"given":"Qianru","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Xidian Universit, Xi&#x2019;an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9905-3238","authenticated-orcid":false,"given":"Zhiwen","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3370-471X","authenticated-orcid":false,"given":"Qing","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Computing, Hong Kong Polytechnic University, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/THMS.2016.2623480"},{"key":"ref2","first-page":"11846","article-title":"Detecting moments and highlights in videos via natural language queries","volume":"34","author":"Lei","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-37731-1_40"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19839-7_17"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/2766966"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201371"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9288"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073653"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/2601097.2601198"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20074-8_12"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3355089.3356520"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3037461"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548268"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00678"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01232"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1207\/s15516709cog1202_4"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1187\/cbe.16-03-0125"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.4324\/9781315208404"},{"volume-title":"An Attentional Theory of Continuity Editing","year":"2006","author":"Smith","key":"ref19"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/2393347.2393373"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2004.826750"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2015.2416554"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-32456-8_39"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00186"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_19"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01058"},{"article-title":"Self-supervised spatiotemporal feature learning via video rotation prediction","year":"2018","author":"Jing","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/THMS.2023.3271625"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref30","first-page":"297","article-title":"Noise-contrastive estimation: A new estimation principle for unnormalized statistical models","volume-title":"Proc. 13th Int. Conf. Artif. Intell. Statist. Workshop Conf.","author":"Gutmann","year":"2010"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01343"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01367"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01065"},{"article-title":"Improving video-text retrieval by multi-stream corpus alignment and dual softmax loss","year":"2021","author":"Cheng","key":"ref35"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531960"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"article-title":"Watching too much television is good: Self-supervised audio-visual representation learning from movies and tv shows","year":"2021","author":"Kalayeh","key":"ref38"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00538"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_41"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3685517"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_2"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"article-title":"Pretext-contrastive learning: Toward good practices in self-supervised video representation leaning","year":"2020","author":"Tao","key":"ref46"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00059"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413694"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1090\/S0002-9939-1968-0223256-9"}],"container-title":["IEEE Transactions on Human-Machine Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6221037\/11199980\/11112832.pdf?arnumber=11112832","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T18:46:29Z","timestamp":1763491589000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11112832\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10]]},"references-count":50,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/thms.2025.3588768","relation":{},"ISSN":["2168-2291","2168-2305"],"issn-type":[{"type":"print","value":"2168-2291"},{"type":"electronic","value":"2168-2305"}],"subject":[],"published":{"date-parts":[[2025,10]]}}}