{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:55:08Z","timestamp":1781816108944,"version":"3.54.5"},"reference-count":60,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"RIE2025 Industry Alignment Fund","award":["I2301E0026"],"award-info":[{"award-number":["I2301E0026"]}]},{"name":"A&#x002A;STAR"},{"name":"Alibaba Group"},{"DOI":"10.13039\/501100001475","name":"Nanyang Technological University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001475","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Nanyang Associate Professorship"},{"name":"National Research Foundation Fellowship","award":["NRFF13-2021-0006"],"award-info":[{"award-number":["NRFF13-2021-0006"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Multimedia"],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/tmm.2026.3654454","type":"journal-article","created":{"date-parts":[[2026,1,15]],"date-time":"2026-01-15T20:50:42Z","timestamp":1768510242000},"page":"4259-4269","source":"Crossref","is-referenced-by-count":0,"title":["Copycat vs. Original: Multi-Modal Pretraining and Variable Importance in Box-Office Prediction"],"prefix":"10.1109","volume":"28","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-7958-1802","authenticated-orcid":false,"given":"Qin","family":"Chao","sequence":"first","affiliation":[{"name":"College of Computing and Data Science, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9332-5653","authenticated-orcid":false,"given":"Eunsoo","family":"Kim","sequence":"additional","affiliation":[{"name":"Business School, University of Seoul, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6230-2376","authenticated-orcid":false,"given":"Boyang","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computing and Data Science, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1088\/1367-2630\/12\/11\/115004"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1111\/joes.12498"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/s10824-019-09372-1"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1287\/mksc.1050.0177"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1086\/381638"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1509\/jmkr.43.2.287"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbusres.2021.03.008"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s11002-011-9146-1"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1086\/508520"},{"key":"ref10","first-page":"34892","article-title":"Visual instruction tuning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Liu","year":"2024"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72983-6_18"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1177\/00222429221127927"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/2492517.2500232"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2016.08.027"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2014.2306681"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1080\/07421222.2016.1243969"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/UBMK.2018.8566661"},{"issue":"14","key":"ref18","first-page":"1","article-title":"What is critical to success in the movie industry? A study on key success factors in the italian motion picture industry","author":"Boccardelli","year":"2008","journal-title":"DIME Work. Papers Intellectual Property Rights"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/EICT.2017.8275242"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1057\/s41272-016-0072-y"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CEC.2018.8477691"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-3414"},{"key":"ref23","first-page":"215","article-title":"Exploiting textual, visual, and product features for predicting the likeability of movies","volume-title":"Proc. 32nd Int. Flairs Conf.","author":"Shafaei","year":"2019"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2010.269"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3002667"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2017.2740022"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2012.2229972"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2016.2614184"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1287\/mksc.17.3.214"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijresmar.2012.04.001"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1287\/isre.2017.0735"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref33","first-page":"13","article-title":"ViLBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lu","year":"2019"},{"key":"ref34","article-title":"VL-BERT: Pre-training of generic visual-linguistic representations","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Su","year":"2020"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"ref36","article-title":"Pixel-BERT: Aligning image pixels with text by deep multi-modal transformers","author":"Huang","year":"2020"},{"key":"ref37","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref38","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li","year":"2022"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/0167-2789(90)90087-6"},{"key":"ref40","first-page":"922","article-title":"Illustrative language understanding: Large-scale visual grounding with image search","volume-title":"Proc. Assoc. Comput. Linguistics","author":"Kiros","year":"2018"},{"key":"ref41","first-page":"2066","article-title":"Vokenization: Improving language understanding with contextualized, visual-grounded supervision","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process.","author":"Tan","year":"2020"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3462244.3479965"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.326"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.78"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.168"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00051"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467338"},{"key":"ref48","first-page":"395","article-title":"A tutorial on spectral clustering","volume-title":"Statist. Comput.","volume":"17","author":"von Luxburg","year":"2007"},{"key":"ref49","article-title":"NumGPT: Improving numeracy ability of generative pre-trained models","volume-title":"Proc. Int. Sympos. Large Lang. Models Finan. Serv. IJCAI","author":"Jin","year":"2023"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.385"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-3020"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1002"},{"key":"ref57","article-title":"Attention interpretability across NLP tasks","author":"Vashishth","year":"2019"},{"key":"ref58","article-title":"Visualizing attention in transformer-based language representation models","author":"Vig","year":"2019"},{"key":"ref59","article-title":"Explainability for vision transformers","author":"Gildenblat","year":"2022"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1201\/9781482269260-7"}],"container-title":["IEEE Transactions on Multimedia"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6046\/11342315\/11353450.pdf?arnumber=11353450","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:11:30Z","timestamp":1781813490000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11353450\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":60,"URL":"https:\/\/doi.org\/10.1109\/tmm.2026.3654454","relation":{},"ISSN":["1520-9210","1941-0077"],"issn-type":[{"value":"1520-9210","type":"print"},{"value":"1941-0077","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}