{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:17:32Z","timestamp":1783315052708,"version":"3.54.6"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T00:00:00Z","timestamp":1776384000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T00:00:00Z","timestamp":1776384000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00530-026-02291-0","type":"journal-article","created":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T11:49:14Z","timestamp":1776426554000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Knowledge enhancement with cross-modal fusion network for radiological report generation"],"prefix":"10.1007","volume":"32","author":[{"given":"Zonglin","family":"Liang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaodi","family":"Hou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangkang","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yijia","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,17]]},"reference":[{"key":"2291_CR1","unstructured":"Jing, B., Xie, P., Xing, E.: On the automatic generation of medical imaging reports. arXiv preprint arXiv:1711.08195 (2017)"},{"issue":"1","key":"2291_CR2","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1007\/s11280-022-01013-6","volume":"26","author":"M Li","year":"2023","unstructured":"Li, M., Liu, R., Wang, F., Chang, X., Liang, X.: Auxiliary signal-guided knowledge encoder-decoder for medical report generation. World Wide Web 26(1), 253\u2013270 (2023)","journal-title":"World Wide Web"},{"key":"2291_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2023.104496","volume":"146","author":"X Hou","year":"2023","unstructured":"Hou, X., Liu, Z., Li, X., Li, X., Sang, S., Zhang, Y.: Mkcl: medical knowledge with contrastive learning model for radiology report generation. J. Biomed. Inform. 146, 104496 (2023)","journal-title":"J. Biomed. Inform."},{"key":"2291_CR4","doi-asserted-by":"crossref","unstructured":"Wang, J., Bhalerao, A., He, Y.: Cross-modal prototype driven network for radiology report generation. In: European Conference on Computer Vision, pp. 563\u2013579 Springer (2022)","DOI":"10.1007\/978-3-031-19833-5_33"},{"key":"2291_CR5","unstructured":"Chen, Z., Shen, Y., Song, Y., Wan, X.: Cross-modal memory networks for radiology report generation. arXiv preprint arXiv:2204.13258 (2022)"},{"key":"2291_CR6","doi-asserted-by":"crossref","unstructured":"Cao, Y., Cui, L., Zhang, L., Yu, F., Li, Z., Xu, Y.: MMTN: multi-modal memory transformer network for image-report consistent medical report generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 277\u2013285 (2023)","DOI":"10.1609\/aaai.v37i1.25100"},{"key":"2291_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.artmed.2020.101878","volume":"106","author":"MMA Monshi","year":"2020","unstructured":"Monshi, M.M.A., Poon, J., Chung, V.: Deep learning in generating radiology reports: a survey. Artif. Intell. Med. 106, 101878 (2020)","journal-title":"Artif. Intell. Med."},{"key":"2291_CR8","doi-asserted-by":"crossref","unstructured":"Chen, Z., Song, Y., Chang, T.-H., Wan, X.: Generating radiology reports via memory-driven transformer. arXiv preprint arXiv:2010.16056 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.112"},{"key":"2291_CR9","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der Maaten, L., Weinberger, K.Q.: Densely connected convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4700\u20134708 (2017)","DOI":"10.1109\/CVPR.2017.243"},{"key":"2291_CR10","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., Adam, H.: Mobilenets: efficient convolutional neural networks for mobile vision applications (2017). arXiv preprint arXiv:1704.04861 126 (2017)"},{"key":"2291_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2291_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106380","volume":"176","author":"K Yang","year":"2024","unstructured":"Yang, K., Li, Q., Tian, C., Zhang, H., Shi, A., Li, J.: Defort: deformable transformer for visual tracking. Neural Netw. 176, 106380 (2024)","journal-title":"Neural Netw."},{"key":"2291_CR13","doi-asserted-by":"publisher","first-page":"1956","DOI":"10.1109\/TMM.2021.3074239","volume":"24","author":"K Yang","year":"2021","unstructured":"Yang, K., He, Z., Pei, W., Zhou, Z., Li, X., Yuan, D., Zhang, H.: Siamcorners: siamese corner networks for visual tracking. IEEE Trans. Multimed. 24, 1956\u20131967 (2021)","journal-title":"IEEE Trans. Multimed."},{"key":"2291_CR14","doi-asserted-by":"crossref","unstructured":"Shin, H.-C., Roberts, K., Lu, L., Demner-Fushman, D., Yao, J., Summers, R.M.: Learning to read chest x-rays: recurrent neural cascade model for automated image annotation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2497\u20132506 (2016)","DOI":"10.1109\/CVPR.2016.274"},{"key":"2291_CR15","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"2291_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2022.102510","volume":"80","author":"S Yang","year":"2022","unstructured":"Yang, S., Wu, X., Ge, S., Zhou, S.K., Xiao, L.: Knowledge matters: chest radiology report generation with general and specific knowledge. Med. Image Anal. 80, 102510 (2022)","journal-title":"Med. Image Anal."},{"key":"2291_CR17","unstructured":"You, J., Li, D., Okumura, M., Suzuki, K.: Jpg-jointly learn to align: automated disease prediction and radiology report generation. In: Proceedings of the 29th International Conference on Computational Linguistics, pp. 5989\u20136001 (2022)"},{"key":"2291_CR18","doi-asserted-by":"crossref","unstructured":"Qin, H., Song, Y.: Reinforced cross-modal alignment for radiology report generation. In: Findings of the Association for Computational Linguistics: ACL 2022, pp. 448\u2013458 (2022)","DOI":"10.18653\/v1\/2022.findings-acl.38"},{"key":"2291_CR19","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F., et al.: Llama open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"2291_CR20","unstructured":"Brown, T.B.: Language models are few-shot learners. arXiv preprint arXiv:2005.14165 (2020)"},{"key":"2291_CR21","unstructured":"Peng, B., Li, C., He, P., Galley, M., Gao, J.: Instruction tuning with gpt-4. arXiv preprint arXiv:2304.03277 (2023)"},{"key":"2291_CR22","unstructured":"Xie, Q., Zhou, J., Peng, Y., Wang, F.: Factreranker: fact-guided reranker for faithful radiology report summarization arXiv preprint arXiv:2303.08335 (2023)"},{"key":"2291_CR23","doi-asserted-by":"crossref","unstructured":"Inan, E.: Making hierarchically aware decisions on short findings for automatic summarisation. J. Comput. Sci. 102692 (2025)","DOI":"10.1016\/j.jocs.2025.102692"},{"key":"2291_CR24","doi-asserted-by":"crossref","unstructured":"Song, L., Xiang, S., Li, F.L.Y.W.Y.: Image captioning: semantic selection unit with stacked residual attention. Image Vision Comput. 144(Apr.), 1\u20131112 (2024)","DOI":"10.1016\/j.imavis.2024.104965"},{"issue":"5","key":"2291_CR25","doi-asserted-by":"publisher","first-page":"4300","DOI":"10.1007\/s10489-024-05389-y","volume":"54","author":"S Du","year":"2024","unstructured":"Du, S., Zhu, H., Lin, G., Liu, Y., Wang, D., Shi, J., Wu, Z.: Weakly supervised grounded image captioning with semantic matching. Appl. Intell. 54(5), 4300\u20134318 (2024)","journal-title":"Appl. Intell."},{"key":"2291_CR26","doi-asserted-by":"crossref","unstructured":"Liu, M., Liu, J., Zhang, X.: Semantic-spatial feature fusion with dynamic graph refinement for remote sensing image captioning (2025)","DOI":"10.1038\/s41598-025-93125-y"},{"issue":"12","key":"2291_CR27","doi-asserted-by":"publisher","first-page":"2724","DOI":"10.1109\/TKDE.2017.2754499","volume":"29","author":"Q Wang","year":"2017","unstructured":"Wang, Q., Mao, Z., Wang, B., Guo, L.: Knowledge graph embedding: a survey of approaches and applications. IEEE Trans. Knowl. Data Eng. 29(12), 2724\u20132743 (2017)","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"2291_CR28","unstructured":"Yang, B., Yih, W.-T., He, X., Gao, J., Deng, L.: Embedding entities and relations for learning and inference in knowledge bases. arXiv preprint arXiv:1412.6575 (2014)"},{"key":"2291_CR29","unstructured":"Trouillon, T., Welbl, J., Riedel, S., Gaussier, \u00c9., Bouchard, G.: Complex embeddings for simple link prediction. In: International Conference on Machine Learning, pp. 2071\u20132080 PMLR (2016)"},{"key":"2291_CR30","doi-asserted-by":"crossref","unstructured":"Dettmers, T., Minervini, P., Stenetorp, P., Riedel, S.: Convolutional 2d knowledge graph embeddings. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.11573"},{"key":"2291_CR31","unstructured":"Sun, Z., Deng, Z.-H., Nie, J.-Y., Tang, J.: Rotate: knowledge graph embedding by relational rotation in complex space. arXiv preprint arXiv:1902.10197 (2019)"},{"issue":"2","key":"2291_CR32","doi-asserted-by":"publisher","first-page":"494","DOI":"10.1109\/TNNLS.2021.3070843","volume":"33","author":"S Ji","year":"2021","unstructured":"Ji, S., Pan, S., Cambria, E., Marttinen, P., Philip, S.Y.: A survey on knowledge graphs: representation, acquisition, and applications. IEEE Trans. Neural Netw. Learn. Syst. 33(2), 494\u2013514 (2021)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"2291_CR33","volume":"24","author":"O Alfarghaly","year":"2021","unstructured":"Alfarghaly, O., Khaled, R., Elkorany, A., Helal, M., Fahmy, A.: Automated radiology report generation using conditioned transformers. Inform. Med. 24, 100557 (2021)","journal-title":"Inform. Med."},{"key":"2291_CR34","doi-asserted-by":"crossref","unstructured":"Jing, B., Wang, Z., Xing, E.: Show, describe and conclude: on exploiting the structure information of chest x-ray reports. arXiv preprint arXiv:2004.12274 (2020)","DOI":"10.18653\/v1\/P19-1657"},{"key":"2291_CR35","doi-asserted-by":"crossref","unstructured":"Shen, H., Pei, M., Liu, J., Tian, Z.: Automatic radiology reports generation via memory alignment network. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 4776\u20134783 (2024)","DOI":"10.1609\/aaai.v38i5.28279"},{"key":"2291_CR36","doi-asserted-by":"crossref","unstructured":"Huang, Z., Zhang, X., Zhang, S.: Kiut: knowledge-injected u-transformer for radiology report generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19809\u201319818 (2023)","DOI":"10.1109\/CVPR52729.2023.01897"},{"key":"2291_CR37","doi-asserted-by":"publisher","first-page":"121260","DOI":"10.1016\/j.eswa.2023.121260","volume":"237","author":"Y Xue","year":"2024","unstructured":"Xue, Y., Tan, Y., Tan, L., Qin, J., Xiang, X.: Generating radiology reports via auxiliary signal guidance and a memory-driven network. Expert Syst. App. 237, 121260 (2024)","journal-title":"Expert Syst. App."},{"issue":"1","key":"2291_CR38","doi-asserted-by":"publisher","first-page":"4542","DOI":"10.1038\/s41467-023-40260-7","volume":"14","author":"X Zhang","year":"2023","unstructured":"Zhang, X., Wu, C., Zhang, Y., Xie, W., Wang, Y.: Knowledge-enhanced visual-language pre-training on chest radiology images. Nat. Commun. 14(1), 4542 (2023)","journal-title":"Nat. Commun."},{"key":"2291_CR39","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der Maaten, L., Weinberger, K.Q.: Densely connected convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4700\u20134708 (2017)","DOI":"10.1109\/CVPR.2017.243"},{"key":"2291_CR40","unstructured":"Tan, M., Le, Q.: Efficientnet: rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114 PMLR (2019)"},{"key":"2291_CR41","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C.: Mobilenetv 2: inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"2291_CR42","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2291_CR43","unstructured":"Gu, A., Dao, T.: Mamba: linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)"},{"key":"2291_CR44","unstructured":"Xu, R., Yang, S., Wang, Y., Du, B., Chen, H.: A survey on vision mamba: models, applications and challenges. arXiv preprint arXiv:2404.18861 (2024)"},{"key":"2291_CR45","unstructured":"Gu, A., Goel, K., R\u00e9, C.: Efficiently modeling long sequences with structured state spaces. arXiv preprint arXiv:2111.00396 (2021)"},{"key":"2291_CR46","doi-asserted-by":"crossref","unstructured":"Zhao, H., Zhang, M., Zhao, W., Ding, P., Huang, S., Wang, D.: Cobra: extending mamba to multi-modal large language model for efficient inference. arXiv preprint arXiv:2403.14520 (2024)","DOI":"10.1609\/aaai.v39i10.33131"},{"key":"2291_CR47","unstructured":"Mao, A., Mohri, M., Zhong, Y.: Cross-entropy loss functions: theoretical analysis and applications. In: International Conference on Machine Learning, pp. 23803\u201323828 PMLR (2023)"},{"key":"2291_CR48","doi-asserted-by":"crossref","unstructured":"Wang, X., Peng, Y., Lu, L., Lu, Z., Bagheri, M., Summers, R.M.: Chestx-ray8: hospital-scale chest X-ray database and benchmarks on weakly-supervised classification and localization of common thorax diseases. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2097\u20132106 (2017)","DOI":"10.1109\/CVPR.2017.369"},{"issue":"1","key":"2291_CR49","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1038\/s41597-019-0322-0","volume":"6","author":"AE Johnson","year":"2019","unstructured":"Johnson, A.E., Pollard, T.J., Berkowitz, S.J., Greenbaum, N.R., Lungren, M.P., Deng, C.-Y., Mark, R.G., Horng, S.: Mimic-cxr, a de-identified publicly available database of chest radiographs with free-text reports. Sci. Data 6(1), 317 (2019)","journal-title":"Sci. Data"},{"key":"2291_CR50","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"2291_CR51","unstructured":"Denkowski, M., Lavie, A.: Meteor 1.3: automatic metric for reliable optimization and evaluation of machine translation systems. In: Proceedings of the Sixth Workshop on Statistical Machine Translation, pp. 85\u201391 (2011)"},{"key":"2291_CR52","unstructured":"Lin, C.-Y.: Rouge: a package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"2291_CR53","unstructured":"Liu, G., Hsu, T.-M.H., McDermott, M., Boag, W., Weng, W.-H., Szolovits, P., Ghassemi, M.: Clinically accurate chest x-ray report generation. In: Machine Learning for Healthcare Conference, pp. 249\u2013269 PMLR(2019)"},{"key":"2291_CR54","doi-asserted-by":"crossref","unstructured":"Wang, X., Peng, Y., Lu, L., Lu, Z., Summers, R.M.: Tienet: text-image embedding network for common thorax disease classification and reporting in chest x-rays. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9049\u20139058 (2018)","DOI":"10.1109\/CVPR.2018.00943"},{"key":"2291_CR55","doi-asserted-by":"crossref","unstructured":"Lu, J., Xiong, C., Parikh, D., Socher, R.: Knowing when to look: adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 375\u2013383 (2017)","DOI":"10.1109\/CVPR.2017.345"},{"key":"2291_CR56","unstructured":"Li, Y., Liang, X., Hu, Z., Xing, E.P.: Hybrid retrieval-generation reinforced agent for medical image report generation. In: Proceedings of the neural information processing systems, pp. 1537\u20131547 (2018)"},{"key":"2291_CR57","doi-asserted-by":"crossref","unstructured":"Liu, F., Wu, X., Ge, S., Fan, W., Zou, Y.: Exploring and distilling posterior and prior knowledge for radiology report generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13753\u201313762 (2021)","DOI":"10.1109\/CVPR46437.2021.01354"},{"key":"2291_CR58","doi-asserted-by":"crossref","unstructured":"Yan, B., Pei, M.: Clinical-bert: vision-language pre-training for radiograph diagnosis and reports generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 2982\u20132990 (2022)","DOI":"10.1609\/aaai.v36i3.20204"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02291-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02291-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02291-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:01:43Z","timestamp":1783314103000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02291-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,17]]},"references-count":58,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2291"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02291-0","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,17]]},"assertion":[{"value":"9 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"233"}}