{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T16:57:15Z","timestamp":1781197035877,"version":"3.54.1"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132107","type":"journal-article","created":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T08:38:58Z","timestamp":1773909538000},"page":"132107","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MAKIMA: Tuning-free multi-attribute open-domain video editing via mask-guided attention modulation"],"prefix":"10.1016","volume":"320","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-3179-6108","authenticated-orcid":false,"given":"Haoyu","family":"Zheng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qifan","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5988-7609","authenticated-orcid":false,"given":"Wenqiao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4889-8118","authenticated-orcid":false,"given":"Hongyang","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2258-1291","authenticated-orcid":false,"given":"Juncheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6529-8088","authenticated-orcid":false,"given":"Zheqi","family":"Lv","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6743-8945","authenticated-orcid":false,"given":"Dongping","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7356-9711","authenticated-orcid":false,"given":"Siliang","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9017-2508","authenticated-orcid":false,"given":"Yueting","family":"Zhuang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132107_bib0001","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6711","article-title":"Restyle: A residual-based StyleGAN encoder via iterative refinement","author":"Alaluf","year":"2021"},{"key":"10.1016\/j.eswa.2026.132107_bib0002","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18208","article-title":"Blended diffusion for text-driven editing of natural images","author":"Avrahami","year":"2022"},{"key":"10.1016\/j.eswa.2026.132107_bib0003","doi-asserted-by":"crossref","unstructured":"Bar-Tal, O., Chefer, H., Tov, O., Herrmann, C., Paiss, R., Zada, S., Ephrat, A., Hur, J., Liu, G., Raj, A., et al., (2024). Lumiere: A space-time diffusion model for video generation. Technical ReportarXiv preprint arXiv: 2401.12945(2024).","DOI":"10.1145\/3680528.3687614"},{"issue":"4","key":"10.1016\/j.eswa.2026.132107_bib0004","doi-asserted-by":"crossref","first-page":"358","DOI":"10.1016\/j.mee.2010.11.019","article-title":"Design and analysis of the in0. 53ga0. 47as implant-free quantum-well device structure","volume":"88","author":"Benbakhti","year":"2011","journal-title":"Microelectronic Engineering"},{"key":"10.1016\/j.eswa.2026.132107_bib0005","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"22560","article-title":"MasaCtrl: Tuning-free mutual self-attention control for consistent image synthesis and editing","author":"Cao","year":"2023"},{"issue":"1","key":"10.1016\/j.eswa.2026.132107_bib0006","doi-asserted-by":"crossref","DOI":"10.1117\/1.3556727","article-title":"Using admittance spectroscopy to quantify transport properties of p3ht thin films","volume":"1","author":"Chan","year":"2011","journal-title":"Journal of Photonics for Energy"},{"issue":"4","key":"10.1016\/j.eswa.2026.132107_bib0007","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3592116","article-title":"Attend-and-excite: Attention-based semantic guidance for text-to-image diffusion models","volume":"42","author":"Chefer","year":"2023","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"10.1016\/j.eswa.2026.132107_bib0008","series-title":"2015 6th International conference on power electronics systems and applications (PESA)","first-page":"1","article-title":"Zero emission electric vessel development","author":"Cheng","year":"2015"},{"key":"10.1016\/j.eswa.2026.132107_bib0009","unstructured":"Cong, Y., Xu, M., Simon, C., Chen, S., Ren, J., Xie, Y., Perez-Rua, J.-M., Rosenhahn, B., Xiang, T., & He, S. (2023). Flatten: Optical flow-guided attention for consistent text-to-video editing. Technical ReportarXiv preprint arXiv: 2310.05922(2023)."},{"issue":"9","key":"10.1016\/j.eswa.2026.132107_bib0010","doi-asserted-by":"crossref","first-page":"10850","DOI":"10.1109\/TPAMI.2023.3261988","article-title":"Diffusion models in vision: A survey","volume":"45","author":"Croitoru","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132107_bib0011","first-page":"16222","article-title":"Diffusion self-guidance for controllable image generation","volume":"36","author":"Epstein","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132107_bib0012","unstructured":"Geyer, M., Bar-Tal, O., Bagon, S., & Dekel, T. (2023). TokenFlow: Consistent diffusion features for consistent video editing. Technical ReportarXiv preprint arXiv: 2307.10373(2023)."},{"key":"10.1016\/j.eswa.2026.132107_bib0013","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., & Cohen-Or, D. (2022). Prompt-to-prompt image editing with cross attention control. Technical ReportarXiv preprint arXiv: 2208.01626(2022)."},{"key":"10.1016\/j.eswa.2026.132107_bib0014","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132107_bib0015","unstructured":"Ho, J., & Salimans, T. (2022). Classifier-free diffusion guidance. Technical ReportarXiv preprint arXiv: 2112.10741(2021)."},{"key":"10.1016\/j.eswa.2026.132107_bib0016","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16750","article-title":"Diffusion-based generation, optimization, and planning in 3D scenes","author":"Huang","year":"2023"},{"issue":"12","key":"10.1016\/j.eswa.2026.132107_bib0017","doi-asserted-by":"crossref","first-page":"5565","DOI":"10.3390\/s23125565","article-title":"Fusion of multi-modal features to enhance dense video caption","volume":"23","author":"Huang","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.eswa.2026.132107_bib0018","unstructured":"Jeong, H., & Ye, J. C. (2023). Ground-a-video: Zero-shot grounded video editing using text-to-image diffusion models. Technical ReportarXiv preprint arXiv: 2310.01107(2023)."},{"key":"10.1016\/j.eswa.2026.132107_bib0019","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15954","article-title":"Text2video-zero: Text-to-image diffusion models are zero-shot video generators","author":"Khachatryan","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_bib0020","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"7701","article-title":"Dense text-to-image generation with attention modulation","author":"Kim","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_bib0021","unstructured":"Kingma, D. P. (2013). Auto-encoding variational bayes. Technical ReportarXiv preprint arXiv: 1312.6114(2013)."},{"key":"10.1016\/j.eswa.2026.132107_bib0022","article-title":"Controllable text-to-image generation","volume":"32","author":"Li","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132107_bib0023","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Jiang, Q., Li, C., Yang, J., Su, H., et al., (2023). Grounding dino: Marrying dino with grounded pre-training for open-set object detection. Technical ReportarXiv preprint arXiv: 2303.05499(2023)."},{"key":"10.1016\/j.eswa.2026.132107_bib0024","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8599","article-title":"Video-p2p: Video editing with cross-attention control","author":"Liu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132107_bib0025","unstructured":"Meng, C., He, Y., Song, Y., Song, J., Wu, J., Zhu, J.-Y., & Ermon, S. (2021). SDEdit: Guided image synthesis and editing with stochastic differential equations. Technical ReportarXiv preprint arXiv: 2108.01073(2021)."},{"key":"10.1016\/j.eswa.2026.132107_bib0026","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"6038","article-title":"Null-text inversion for editing real images using guided diffusion models","author":"Mokady","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_bib0027","unstructured":"Nichol, A., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., Mcgrew, B., Sutskever, I., & Chen, M. (2021). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. Technical ReportarXiv preprint arXiv: 2112.10741(2021)."},{"key":"10.1016\/j.eswa.2026.132107_bib0028","series-title":"ACM SIGGRAPH 2023 conference proceedings","first-page":"1","article-title":"Zero-shot image-to-image translation","author":"Parmar","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_bib0029","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"724","article-title":"A benchmark dataset and evaluation methodology for video object segmentation","author":"Perazzi","year":"2016"},{"key":"10.1016\/j.eswa.2026.132107_bib0030","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15932","article-title":"Fatezero: Fusing attentions for zero-shot text-based video editing","author":"Qi","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_bib0031","series-title":"International conference on machine learning, PMLR","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132107_bib0032","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv: 2204.06125 1 (2) (2022) 3."},{"key":"10.1016\/j.eswa.2026.132107_bib0033","series-title":"International conference on machine learning, PMLR","first-page":"8821","article-title":"Zero-shot text-to-image generation","author":"Ramesh","year":"2021"},{"key":"10.1016\/j.eswa.2026.132107_bib0034","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., & Sutskever, I. (2021b). Zero-shot text-to-image generation. https:\/\/arxiv.org\/abs\/2102.12092."},{"key":"10.1016\/j.eswa.2026.132107_bib0035","unstructured":"Ravi, N., Gabeur, V., Hu, Y.-T., Hu, R., Ryali, C., Ma, T., Khedr, H., R\u00e4dle, R., Rolland, C., & Gustafson, L., et al., (2024). Sam2: Segment anything in images and videos. Technical ReportarXiv preprint arXiv: 2408.00714(2024)."},{"key":"10.1016\/j.eswa.2026.132107_bib0036","series-title":"International conference on machine learning, PMLR","first-page":"1060","article-title":"Generative adversarial text to image synthesis","author":"Reed","year":"2016"},{"key":"10.1016\/j.eswa.2026.132107_bib0037","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10684","article-title":"High-resolution image synthesis with latent diffusion models","author":"Rombach","year":"2022"},{"key":"10.1016\/j.eswa.2026.132107_bib0038","unstructured":"Song, J., Meng, C., & Ermon, S. (2020). Denoising diffusion implicit models. Technical ReportarXiv preprint arXiv: 2010.02502(2020)."},{"key":"10.1016\/j.eswa.2026.132107_bib0039","article-title":"Any-to-any generation via composable diffusion","volume":"36","author":"Tang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132107_bib0040","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1921","article-title":"Plug-and-play diffusion features for text-driven image-to-image translation","author":"Tumanyan","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_bib0041","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"7623","article-title":"Tune-a-video: One-shot tuning of image diffusion models for text-to-video generation","author":"Wu","year":"2023"},{"issue":"4","key":"10.1016\/j.eswa.2026.132107_bib0042","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3626235","article-title":"Diffusion models: A comprehensive survey of methods and applications","volume":"56","author":"Yang","year":"2023","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.132107_bib0043","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8703","article-title":"Fresco: Spatial-temporal correspondence for zero-shot video translation","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132107_bib0044","series-title":"Proceedings of the IEEE international conference on computer vision and pattern recognition","first-page":"5907","article-title":"StackGAN: Text to photo-realistic image synthesis with stacked generative adversarial networks","author":"Zhang","year":"2017"},{"key":"10.1016\/j.eswa.2026.132107_bib0045","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3836","article-title":"Adding conditional control to text-to-image diffusion models","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132107_sbref0046","series-title":"Technical Report","article-title":"ControlVideo: Training-free controllable text-to-video generation","author":"Zhang","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010201?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010201?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T15:58:44Z","timestamp":1781193524000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426010201"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":46,"alternative-id":["S0957417426010201"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132107","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MAKIMA: Tuning-free multi-attribute open-domain video editing via mask-guided attention modulation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132107","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"132107"}}