{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T22:42:08Z","timestamp":1785278528453,"version":"3.55.0"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T00:00:00Z","timestamp":1742256000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T00:00:00Z","timestamp":1742256000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LQ23F010005"],"award-info":[{"award-number":["LQ23F010005"]}]},{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LY24F030012"],"award-info":[{"award-number":["LY24F030012"]}]},{"name":"Scientific Research Fund of Zhejiang ProvincialEducation Department","award":["Y202250022"],"award-info":[{"award-number":["Y202250022"]}]},{"name":"Joint Funds of the Zhejiang Provincial NaturalScience Foundation of China","award":["LHY21E090004"],"award-info":[{"award-number":["LHY21E090004"]}]},{"name":"\"The Professional Development Projects of Teachers\" for Domestic Visiting Scholars of Colleges and Universities in 2022, China","award":["FX2022075"],"award-info":[{"award-number":["FX2022075"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-07111-2","type":"journal-article","created":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T15:34:02Z","timestamp":1742312042000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Optimizing medical image report generation through a discrete diffusion framework"],"prefix":"10.1007","volume":"81","author":[{"given":"Shuifa","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhanglin","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junsen","family":"Meizhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiacheng","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keyong","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,18]]},"reference":[{"issue":"1","key":"7111_CR1","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1038\/s41597-019-0322-0","volume":"6","author":"AE Johnson","year":"2019","unstructured":"Johnson AE, Pollard TJ, Berkowitz SJ, Greenbaum NR, Lungren MP, Deng C-Y, Mark RG, Horng S (2019) Mimic-cxr, a de-identified publicly available database of chest radiographs with free-text reports. Sci Data 6(1):317","journal-title":"Sci Data"},{"issue":"11","key":"7111_CR2","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2020) Generative adversarial networks. Commun ACM 63(11):139\u2013144","journal-title":"Commun ACM"},{"issue":"6","key":"7111_CR3","doi-asserted-by":"publisher","first-page":"689","DOI":"10.1016\/0378-2166(87)90109-3","volume":"11","author":"E Hovy","year":"1987","unstructured":"Hovy E (1987) Generating natural language under pragmatic constraints. J Pragmat 11(6):689\u2013719","journal-title":"J Pragmat"},{"key":"7111_CR4","unstructured":"Sutskever I, Vinyals O, Le QV (2014) Sequence to sequence learning with neural networks. In: Proceedings of the 28th International Conference on Neural Information Processing Systems - Volume 2. NIPS\u201914, pp. 3104\u20133112. MIT Press, Cambridge, MA, USA"},{"key":"7111_CR5","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, Erhan D (2015) Show and tell: A neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"7111_CR6","doi-asserted-by":"crossref","unstructured":"Jing B, Xie P, Xing E (2018) On the automatic generation of medical imaging reports. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2577\u20132586","DOI":"10.18653\/v1\/P18-1240"},{"key":"7111_CR7","doi-asserted-by":"crossref","unstructured":"Yin C, Qian B, Wei J, Li X, Zhang X, Li Y, Zheng Q (2019) Automatic generation of medical imaging diagnostic report with hierarchical recurrent neural network. In: 2019 IEEE International Conference on Data Mining (ICDM), pp. 728\u2013737","DOI":"10.1109\/ICDM.2019.00083"},{"key":"7111_CR8","doi-asserted-by":"crossref","unstructured":"Kong M, Huang Z, Kuang K, Zhu Q, Wu F (2022) Transq: Transformer-based semantic query for medical report generation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 610\u2013620","DOI":"10.1007\/978-3-031-16452-1_58"},{"key":"7111_CR9","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho J, Jain A, Abbeel P (2020) Denoising diffusion probabilistic models. Adv Neural Inf Process Syst 33:6840\u20136851","journal-title":"Adv Neural Inf Process Syst"},{"key":"7111_CR10","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal P, Nichol A (2021) Diffusion models beat gans on image synthesis. Adv Neural Inf Process Syst 34:8780\u20138794","journal-title":"Adv Neural Inf Process Syst"},{"key":"7111_CR11","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Proceedings of the 31st International Conference on Neural Information Processing Systems. NIPS\u201917, pp. 6000\u20136010. Curran Associates Inc., Red Hook, NY, USA"},{"issue":"4","key":"7111_CR12","doi-asserted-by":"publisher","first-page":"2152","DOI":"10.1109\/JBHI.2024.3350077","volume":"28","author":"X Yi","year":"2024","unstructured":"Yi X, Fu Y, Liu R, Zhang H, Hua R (2024) Tsget: Two-stage global enhanced transformer for automatic radiology report generation. IEEE J Biomed Health Inform 28(4):2152\u20132162","journal-title":"IEEE J Biomed Health Inform"},{"key":"7111_CR13","unstructured":"Yi X, Fu Y, Yu J, Liu R, Zhang H, Hua R (2024) Lhr-rfl: Linear hybrid-reward based reinforced focal learning for automatic radiology report generation. IEEE Trans Med Imaging, 1\u20131"},{"key":"7111_CR14","doi-asserted-by":"crossref","unstructured":"Yao Z, Lin F, Chai S, He W, Dai L, Fei X (2024) Integrating medical imaging and clinical reports using multimodal deep learning for advanced disease analysis. In: 2024 IEEE 2nd International Conference on Sensors, Electronics and Computer Engineering (ICSECE), pp. 1217\u20131223","DOI":"10.1109\/ICSECE61636.2024.10729527"},{"key":"7111_CR15","doi-asserted-by":"crossref","unstructured":"Song S, Li X, Li S, Zhao S, Yu J, Ma J, Mao X, Zhang W, Wang M (2025) How to bridge the gap between modalities: survey on multimodal large language model. IEEE Trans Knowl Data Eng 1\u201320","DOI":"10.1109\/TKDE.2025.3527978"},{"issue":"1","key":"7111_CR16","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1186\/s12938-023-01113-y","volume":"22","author":"T Pang","year":"2023","unstructured":"Pang T, Li P, Zhao L (2023) A survey on automatic generation of medical imaging reports based on deep learning. BioMed Eng OnLine 22(1):48","journal-title":"BioMed Eng OnLine"},{"issue":"1","key":"7111_CR17","first-page":"60","volume":"41","author":"S Xing","year":"2024","unstructured":"Xing S, Fang J, Ju Z, Guo Z, Wang Y (2024) Research on automatic generation of multimodal medical image reports based on memory driven. J Biomed Eng 41(1):60\u201369","journal-title":"J Biomed Eng"},{"issue":"4","key":"7111_CR18","doi-asserted-by":"publisher","first-page":"2199","DOI":"10.1109\/JBHI.2024.3354712","volume":"28","author":"J Wang","year":"2024","unstructured":"Wang J, Bhalerao A, Yin T, See S, He Y (2024) Camanet: Class activation map guided attention network for radiology report generation. IEEE J Biomed Health Inform 28(4):2199\u20132210","journal-title":"IEEE J Biomed Health Inform"},{"key":"7111_CR19","unstructured":"Ranjit M, Ganapathy G, Manuel R, Ganu T (2023) Retrieval augmented chest x-ray report generation using openai gpt models. In: Machine Learning for Healthcare Conference, pp. 650\u2013666"},{"key":"7111_CR20","unstructured":"Nichol AQ, Dhariwal P (2021) Improved denoising diffusion probabilistic models. In: International Conference on Machine Learning, pp. 8162\u20138171"},{"key":"7111_CR21","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia C, Chan W, Saxena S, Li L, Whang J, Denton EL, Ghasemipour K, Gontijo Lopes R, Karagol Ayan B, Salimans T (2022) Photorealistic text-to-image diffusion models with deep language understanding. Adv Neural Inf Process Syst 35:36479\u201336494","journal-title":"Adv Neural Inf Process Syst"},{"key":"7111_CR22","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"7111_CR23","doi-asserted-by":"crossref","unstructured":"Ronneberger O, Fischer P, Brox T (2015) U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-assisted intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, October 5-9, 2015, Proceedings, Part III 18, pp. 234\u2013241","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"7111_CR24","doi-asserted-by":"crossref","unstructured":"Rennie SJ, Marcheret E, Mroueh Y, Ross J, Goel V (2017) Self-critical sequence training for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7008\u20137024","DOI":"10.1109\/CVPR.2017.131"},{"key":"7111_CR25","doi-asserted-by":"crossref","unstructured":"Lu J, Xiong C, Parikh D, Socher R (2017) Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 375\u2013383","DOI":"10.1109\/CVPR.2017.345"},{"key":"7111_CR26","doi-asserted-by":"crossref","unstructured":"Chen Z, Shen Y, Song Y, Wan X (2021) Cross-modal memory networks for radiology report generation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 5904\u20135914","DOI":"10.18653\/v1\/2021.acl-long.459"},{"key":"7111_CR27","doi-asserted-by":"crossref","unstructured":"Chen Z, Song Y, Chang T-H, Wan X (2020) Generating radiology reports via memory-driven transformer. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), 1439\u20131449","DOI":"10.18653\/v1\/2020.emnlp-main.112"},{"key":"7111_CR28","doi-asserted-by":"crossref","unstructured":"Liu F, Ge S, Wu X (2021) Competence-based multimodal curriculum learning for medical report generation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 3001\u20133012","DOI":"10.18653\/v1\/2021.acl-long.234"},{"key":"7111_CR29","unstructured":"Li CY, Liang X, Hu Z, Xing EP (2018) Hybrid retrieval-generation reinforced agent for medical image report generation. In: Proceedings of the 32nd International Conference on Neural Information Processing Systems, pp. 1537\u20131547"},{"key":"7111_CR30","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2023.102798","volume":"86","author":"S Yang","year":"2023","unstructured":"Yang S, Wu X, Ge S, Zheng Z, Zhou SK, Xiao L (2023) Radiology report generation with a learned knowledge base and multi-modal alignment. Med Image Anal 86:102798","journal-title":"Med Image Anal"},{"key":"7111_CR31","doi-asserted-by":"crossref","unstructured":"Tanida T, M\u00fcller P, Kaissis G, Rueckert D (2023) Interactive and explainable region-guided radiology report generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7433\u20137442","DOI":"10.1109\/CVPR52729.2023.00718"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07111-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-07111-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07111-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T15:34:20Z","timestamp":1742312060000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-07111-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,18]]},"references-count":31,"journal-issue":{"issue":"5","published-online":{"date-parts":[[2025,4]]}},"alternative-id":["7111"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-07111-2","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,18]]},"assertion":[{"value":"21 February 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"637"}}