{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T04:27:35Z","timestamp":1786595255512,"version":"build-2736575974"},"reference-count":47,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Applied Soft Computing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.asoc.2026.115840","type":"journal-article","created":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T23:06:33Z","timestamp":1782601593000},"page":"115840","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["A scene-aware soft multimodal transformer with controllable alignment for interactive narrative generation"],"prefix":"10.1016","volume":"202","author":[{"given":"Siyu","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinjin","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manzhou","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2428-9211","authenticated-orcid":false,"given":"Yan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.asoc.2026.115840_bib0005","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"6267","article-title":"Generative AI for film creation: a survey of recent advances","author":"Zhang","year":"2025"},{"issue":"7","key":"10.1016\/j.asoc.2026.115840_bib0010","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3736724","article-title":"Innovations and challenges of AI in film: a methodological framework for future exploration","volume":"21","author":"Uddin","year":"2025","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"3","key":"10.1016\/j.asoc.2026.115840_bib0015","article-title":"Interactive movies: narrative and immersive experience","volume":"7","author":"Hang","year":"2024","journal-title":"Acad. J. Humanit. Soc. Sci."},{"key":"10.1016\/j.asoc.2026.115840_bib0020","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.asoc.2026.115840_bib0025","series-title":"Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume","first-page":"300","article-title":"Recipes for building an open-domain chatbot","author":"Roller","year":"2021"},{"key":"10.1016\/j.asoc.2026.115840_bib0030","series-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics: System Demonstrations","first-page":"270","article-title":"Dialogpt: large-scale generative pre-training for conversational response generation","author":"Zhang","year":"2020"},{"issue":"1","key":"10.1016\/j.asoc.2026.115840_bib0035","first-page":"67","article-title":"Interactive narrative: an intelligent systems approach","volume":"34","author":"Riedl","year":"2013","journal-title":"AI Mag."},{"key":"10.1016\/j.asoc.2026.115840_bib0040","author":"Liang"},{"key":"10.1016\/j.asoc.2026.115840_bib0045","doi-asserted-by":"crossref","DOI":"10.3389\/frobt.2025.1662819","article-title":"Exploring multimodal collaborative storytelling with pepper: a preliminary study with zero-shot LLMs","volume":"12","author":"Zabala","year":"2025","journal-title":"Front. Robot. AI"},{"key":"10.1016\/j.asoc.2026.115840_bib0050","doi-asserted-by":"crossref","DOI":"10.3389\/frvir.2022.854960","article-title":"Interactive digital narratives as complex expressive means","volume":"3","author":"Bellini","year":"2022","journal-title":"Front. Virtual Real."},{"key":"10.1016\/j.asoc.2026.115840_bib0055","author":"Achiam"},{"key":"10.1016\/j.asoc.2026.115840_bib0060","series-title":"ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"2352","article-title":"End-to-end audio visual scene-aware dialog using multimodal attention-based video features","author":"Hori","year":"2019"},{"key":"10.1016\/j.asoc.2026.115840_bib0065","series-title":"European Conference on Computer Vision","first-page":"709","article-title":"Movienet: a holistic dataset for movie understanding","author":"Huang","year":"2020"},{"key":"10.1016\/j.asoc.2026.115840_bib0070","doi-asserted-by":"crossref","DOI":"10.3389\/fcomm.2024.1347788","article-title":"Multimodal cohesion and viewers\u2019 comprehension of scene transitions in film: an empirical investigation","volume":"9","author":"Markhabayeva","year":"2024","journal-title":"Front. Commun."},{"issue":"1","key":"10.1016\/j.asoc.2026.115840_bib0075","doi-asserted-by":"crossref","DOI":"10.1080\/08839514.2024.2419609","article-title":"Enhancing multimodal emotional information extraction in film and television through adaptive feature fusion with DenseNe, transformer, and 3D CNN models","volume":"38","author":"Liang","year":"2024","journal-title":"Appl. Artif. Intell."},{"issue":"3","key":"10.1016\/j.asoc.2026.115840_bib0080","doi-asserted-by":"crossref","first-page":"796","DOI":"10.1177\/10776990241257586","article-title":"Effects of visual framing in multimodal media environments: a systematic review of studies between 1979 and 2023","volume":"102","author":"Geise","year":"2025","journal-title":"Journal. Mass Commun. Q."},{"key":"10.1016\/j.asoc.2026.115840_bib0085","doi-asserted-by":"crossref","DOI":"10.1016\/j.dcm.2021.100539","article-title":"Cut to the chase\u2013how multimodal cohesion secures narrative orientation in film trailers","volume":"44","author":"Hoffmann","year":"2021","journal-title":"Discourse Context Media"},{"key":"10.1016\/j.asoc.2026.115840_bib0090","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7558","article-title":"Audio visual scene-aware dialog","author":"Alamri","year":"2019"},{"issue":"1","key":"10.1016\/j.asoc.2026.115840_bib0095","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1057\/s41599-025-04510-x","article-title":"Mapping the knowledge domain of multimodal translation: a bibliometric analysis","volume":"12","author":"Guo","year":"2025","journal-title":"Humanit. Soc. Sci. Commun."},{"key":"10.1016\/j.asoc.2026.115840_bib0100","doi-asserted-by":"crossref","DOI":"10.1016\/j.displa.2024.102708","article-title":"Visual-guided scene-aware audio generation method based on hierarchical feature codec and rendering decision","volume":"83","author":"Wang","year":"2024","journal-title":"Displays"},{"key":"10.1016\/j.asoc.2026.115840_bib0105","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"13392","article-title":"Long-range multimodal pretraining for movie understanding","author":"Argaw","year":"2023"},{"key":"10.1016\/j.asoc.2026.115840_bib0110","series-title":"European Conference on Computer Vision","first-page":"375","article-title":"Learning video context as interleaved multimodal sequences","author":"Lin","year":"2024"},{"issue":"1","key":"10.1016\/j.asoc.2026.115840_bib0115","doi-asserted-by":"crossref","first-page":"40","DOI":"10.1080\/10926488.2023.2271544","article-title":"Identifying and interpreting visual and multimodal metaphor in commercials and feature films","volume":"39","author":"Forceville","year":"2024","journal-title":"Metaphor Symb."},{"key":"10.1016\/j.asoc.2026.115840_bib0120","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.asoc.2026.115840_bib0125","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"7464","article-title":"Videobert: a joint model for video and language representation learning","author":"Sun","year":"2019"},{"key":"10.1016\/j.asoc.2026.115840_bib0130","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8746","article-title":"Actbert: learning global-local video-text representations","author":"Zhu","year":"2020"},{"key":"10.1016\/j.asoc.2026.115840_bib0135","first-page":"23634","article-title":"Merlot: multimodal neural script knowledge models","volume":"34","author":"Zellers","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.asoc.2026.115840_bib0140","author":"Fu"},{"key":"10.1016\/j.asoc.2026.115840_bib0145","author":"Xue"},{"key":"10.1016\/j.asoc.2026.115840_bib0150","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15180","article-title":"Imagebind: one embedding space to bind them all","author":"Girdhar","year":"2023"},{"key":"10.1016\/j.asoc.2026.115840_bib0155","author":"Li"},{"key":"10.1016\/j.asoc.2026.115840_bib0160","author":"Zhang"},{"key":"10.1016\/j.asoc.2026.115840_bib0165","author":"Gao"},{"key":"10.1016\/j.asoc.2026.115840_bib0170","first-page":"14200","article-title":"Attention bottlenecks for multimodal fusion","volume":"34","author":"Nagrani","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.asoc.2026.115840_bib0175","author":"Hong"},{"key":"10.1016\/j.asoc.2026.115840_bib0180","series-title":"Proceedings of the Conference. Association for Computational Linguistics. Meeting","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","author":"Tsai","year":"2019"},{"issue":"10","key":"10.1016\/j.asoc.2026.115840_bib0185","doi-asserted-by":"crossref","first-page":"12113","DOI":"10.1109\/TPAMI.2023.3275156","article-title":"Multimodal learning with transformers: a survey","volume":"45","author":"Xu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"10.1016\/j.asoc.2026.115840_bib0190","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1016\/j.entcom.2012.09.004","article-title":"Experiencing interactive narrative: a qualitative analysis of fa\u00e7ade","volume":"4","author":"El-Nasr","year":"2013","journal-title":"Entertain. Comput."},{"key":"10.1016\/j.asoc.2026.115840_bib0195","author":"Clerc"},{"key":"10.1016\/j.asoc.2026.115840_bib0200","doi-asserted-by":"crossref","DOI":"10.3389\/frobt.2021.716581","article-title":"Personal narratives in technology design: the value of sharing older adults\u2019 stories in the design of social robots","volume":"8","author":"Ostrowski","year":"2021","journal-title":"Front. Robot. AI"},{"issue":"9","key":"10.1016\/j.asoc.2026.115840_bib0205","doi-asserted-by":"crossref","first-page":"471","DOI":"10.3390\/systems11090471","article-title":"How does interactive narrative design affect the consumer experience of mobile interactive video advertising?","volume":"11","author":"Gu","year":"2023","journal-title":"Systems"},{"key":"10.1016\/j.asoc.2026.115840_bib0210","author":"Nack"},{"key":"10.1016\/j.asoc.2026.115840_bib0215","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1016\/j.dr.2014.12.004","article-title":"Affordances and limitations of electronic storybooks for young children\u2019s emergent literacy","volume":"35","author":"Bus","year":"2015","journal-title":"Dev. Rev."},{"issue":"4","key":"10.1016\/j.asoc.2026.115840_bib0220","doi-asserted-by":"crossref","first-page":"698","DOI":"10.3102\/0034654314566989","article-title":"Benefits and pitfalls of multimedia and interactive features in technology-enhanced storybooks: a meta-analysis","volume":"85","author":"Takacs","year":"2015","journal-title":"Rev. Educ. Res."},{"issue":"4","key":"10.1016\/j.asoc.2026.115840_bib0225","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/1412196.1412198","article-title":"Interactive TV narratives: opportunities, progress, and challenges","volume":"4","author":"Ursu","year":"2008","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl. (TOMM)"},{"issue":"3\u20134","key":"10.1016\/j.asoc.2026.115840_bib0230","doi-asserted-by":"crossref","first-page":"76","DOI":"10.1080\/13614568.2023.2181503","article-title":"Interactive digital narrative (IDN)\u2014new ways to represent complexity and facilitate digitally empowered citizens","volume":"28","author":"Koenitz","year":"2022","journal-title":"New Rev. Hypermedia Multimed."},{"key":"10.1016\/j.asoc.2026.115840_bib0235","author":"Li"}],"container-title":["Applied Soft Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1568494626012883?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1568494626012883?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,13]],"date-time":"2026-08-13T03:40:31Z","timestamp":1786592431000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1568494626012883"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":47,"alternative-id":["S1568494626012883"],"URL":"https:\/\/doi.org\/10.1016\/j.asoc.2026.115840","relation":{},"ISSN":["1568-4946"],"issn-type":[{"value":"1568-4946","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A scene-aware soft multimodal transformer with controllable alignment for interactive narrative generation","name":"articletitle","label":"Article Title"},{"value":"Applied Soft Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.asoc.2026.115840","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"115840"}}