{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T00:30:56Z","timestamp":1742949056703,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":37,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819794362"},{"type":"electronic","value":"9789819794379"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-9437-9_36","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T16:29:25Z","timestamp":1730392165000},"page":"458-471","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-granularity Semantic Guided Transformer for Radiology Report Generation"],"prefix":"10.1007","author":[{"given":"Yu","family":"Song","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaojin","family":"Hua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kunli","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongying","family":"Zan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Runzhi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"36_CR1","doi-asserted-by":"crossref","unstructured":"Zeng, P., Zhang, H., Song, J., Gao, L.: S2 transformer for image captioning. In: IJCAI, pp. 1608\u20131614 (2022)","DOI":"10.24963\/ijcai.2022\/224"},{"key":"36_CR2","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10578\u201310587 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"36_CR3","unstructured":"Ma, X., et al.: Image as set of points. In: The Eleventh International Conference on Learning Representations (2023)"},{"key":"36_CR4","unstructured":"Han, K., Wang, Y., Guo, J., Tang, Y., Wu, E.: Vision GNN: an image is worth graph of nodes. In: Neural Information Processing Systems, vol. 35, pp. 8291\u20138303 (2022)"},{"key":"36_CR5","doi-asserted-by":"crossref","unstructured":"Lin, Y., Chen, M., Zhang, K., et al.: TagCLIP: a local-to-global framework to enhance open-vocabulary multi-label classification of CLIP without training. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 4, pp. 3513\u20133521 (2024)","DOI":"10.1609\/aaai.v38i4.28139"},{"key":"36_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Z., Song, Y., Chang, T.H., Wan, X.: Generating radiology reports via memory-driven transformer. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing, EMNLP 2020, pp. 1439\u20131449 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.112"},{"key":"36_CR7","doi-asserted-by":"publisher","unstructured":"Nguyen, V. Q., Suganuma, M., Okatani, T.: GRIT: faster and better image captioning transformer using dual visual features. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, pp. 167\u2013184. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_10","DOI":"10.1007\/978-3-031-20059-5_10"},{"key":"36_CR8","doi-asserted-by":"crossref","unstructured":"Luo, Y., Ji, J., Sun, X., et al.: Dual-level collaborative transformer for image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, no. 3, pp. 2286\u20132293 (2021)","DOI":"10.1609\/aaai.v35i3.16328"},{"key":"36_CR9","doi-asserted-by":"crossref","unstructured":"Chen, Z., Shen, Y., Song, Y., Wan, X.: Cross-modal memory networks for radiology report generation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, Volume 1: Long Papers, pp. 5904\u20135914 (2021)","DOI":"10.18653\/v1\/2021.acl-long.459"},{"key":"36_CR10","unstructured":"Ren, S., He, K., Girshick, R.B., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Proceedings of NeurIPS (2015)"},{"key":"36_CR11","doi-asserted-by":"crossref","unstructured":"Jiang, H., Misra, I., Rohrbach, M., Learned-Miller, E., Chen, X.: In defense of grid features for visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10267\u201310276 (2020)","DOI":"10.1109\/CVPR42600.2020.01028"},{"key":"36_CR12","doi-asserted-by":"crossref","unstructured":"Wang, Z., Liu, L., Wang, L., Zhou, L.: Metransformer: radiology report generation by transformer with multiple learnable expert tokens. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11558\u201311567 (2023)","DOI":"10.1109\/CVPR52729.2023.01112"},{"key":"36_CR13","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"36_CR14","doi-asserted-by":"crossref","unstructured":"Tanida, T., M\u00fcller, P., Kaissis, G., Rueckert, D.: Interactive and explainable region-guided radiology report generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7433\u20137442 (2023)","DOI":"10.1109\/CVPR52729.2023.00718"},{"key":"36_CR15","doi-asserted-by":"crossref","unstructured":"Huang, Z., Zhang, X., Zhang, S.: KIUT: knowledge-injected u-transformer for radiology report generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19809\u201319818 (2023)","DOI":"10.1109\/CVPR52729.2023.01897"},{"key":"36_CR16","unstructured":"Ashish, V., Noam, S., Niki, P., et al.: Attention is all you need. In: Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"key":"36_CR17","doi-asserted-by":"publisher","unstructured":"Hou, X., Liu, Z., Li, X., Li, X., Sang, S., Zhang, Y.: MKCL: medical knowledge with contrastive learning model for radiology report generation. J. Biomed. Inform. 104496 (2023). https:\/\/doi.org\/10.1016\/j.jbi.2023.104496","DOI":"10.1016\/j.jbi.2023.104496"},{"key":"36_CR18","doi-asserted-by":"crossref","unstructured":"Liu, F., Wu, X., Ge, S., Fan, W., Zou, Y.: Exploring and distilling posterior and prior knowledge for radiology report generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13753\u201313762 (2021)","DOI":"10.1109\/CVPR46437.2021.01354"},{"issue":"5","key":"36_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3326362","volume":"38","author":"Y Wang","year":"2019","unstructured":"Wang, Y., Sun, Y., Liu, Z., Sarma, S.E., et al.: Dynamic graph CNN for learning on point clouds. ACM Trans. Graphics (tog) 38(5), 1\u201312 (2019)","journal-title":"ACM Trans. Graphics (tog)"},{"key":"36_CR20","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xu, J., & Sun, Y.: End-to-end transformer based model for image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, No. 3, pp. 2585\u20132594 (2022)","DOI":"10.1609\/aaai.v36i3.20160"},{"key":"36_CR21","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"36_CR22","doi-asserted-by":"publisher","unstructured":"Graves, A., Graves, A.: Long short-term memory. In: Graves, A. (ed.) Supervised Sequence Labelling with Recurrent Neural Networks, pp. 37\u201345. Springer, Heidelberg (2012). https:\/\/doi.org\/10.1007\/978-3-642-24797-2_4","DOI":"10.1007\/978-3-642-24797-2_4"},{"issue":"9","key":"36_CR23","doi-asserted-by":"publisher","first-page":"3786","DOI":"10.1109\/TNNLS.2021.3099165","volume":"32","author":"G Liu","year":"2021","unstructured":"Liu, G., Liao, Y., Wang, F., Zhang, B., et al.: Medical-VLBERT: medical visual language BERT for COVID-19 CT report generation with alternate learning. IEEE Trans. Neural Networks Learn. Syst. 32(9), 3786\u20133797 (2021)","journal-title":"IEEE Trans. Neural Networks Learn. Syst."},{"issue":"1","key":"36_CR24","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1007\/s11280-022-01013-6","volume":"26","author":"M Li","year":"2022","unstructured":"Li, M., Liu, R., Wang, F., Chang, X., Liang, X.: Auxiliary signal-guided knowledge encoder-decoder for medical report generation. World Wide web 26(1), 253\u2013270 (2022)","journal-title":"World Wide web"},{"issue":"2","key":"36_CR25","doi-asserted-by":"publisher","first-page":"304","DOI":"10.1093\/jamia\/ocv080","volume":"23","author":"D Demner-Fushman","year":"2016","unstructured":"Demner-Fushman, D., Kohli, M.D., Rosenman, M.B., Shooshan, S.E., et al.: Preparing a collection of radiology examinations for distribution and retrieval. J. Am. Med. Inform. Assoc. 23(2), 304\u2013310 (2016)","journal-title":"J. Am. Med. Inform. Assoc."},{"key":"36_CR26","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"36_CR27","unstructured":"Banerjee, S., Lavie, A.: Meteor: an automatic metric for MT evaluation with improved correlation with human judgments. In: IEEvaluation@ACL (2005)"},{"key":"36_CR28","unstructured":"Lin, C.Y.: Rouge: a package for automatic evaluation of summaries. In: Annual Meeting of the Association for Computational Linguistics (2004)"},{"key":"36_CR29","doi-asserted-by":"crossref","unstructured":"Jing, B., Xie, P., & Xing, E.: On the automatic generation of medical imaging reports. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics, Volume 1: Long Papers, pp. 2577\u20132586 (2018)","DOI":"10.18653\/v1\/P18-1240"},{"issue":"1","key":"36_CR30","first-page":"6680546","volume":"2024","author":"Y Tan","year":"2024","unstructured":"Tan, Y., Li, C., Qin, J., Xue, Y., Xiang, X.: Medical image description based on multimodal auxiliary signals and transformer. Int. J. Intell. Syst. 2024(1), 6680546 (2024)","journal-title":"Int. J. Intell. Syst."},{"key":"36_CR31","doi-asserted-by":"crossref","unstructured":"Lu, J., Xiong, C., Parikh, D., Socher, R.: Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 375\u2013383 (2017)","DOI":"10.1109\/CVPR.2017.345"},{"key":"36_CR32","unstructured":"Team GLM, Zeng, A., Xu, B., Wang, B., et al.: ChatGLM: a family of large language models from GLM-130B to GLM-4 all tools. arXiv preprint arXiv:2406.12793 (2024)"},{"key":"36_CR33","unstructured":"Devlin, J., Chang, M., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: NAACL 2019, pp. 4171\u20134186 (2019)"},{"key":"36_CR34","doi-asserted-by":"publisher","unstructured":"You, D., Liu, F., Ge, S., Xie, X., Zhang, J., Wu, X.: Aligntransformer: hierarchical alignment of visual regions and disease tags for medical report generation. In: de Bruijne, M., et al. (eds.) MICCAI 2021. LNCS, vol. 12903, pp. 72\u201382. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87199-4_7","DOI":"10.1007\/978-3-030-87199-4_7"},{"key":"36_CR35","doi-asserted-by":"crossref","unstructured":"Johnson, A.E., Pollard, T.J., Greenbaum, N.R., Lungren, M.P., et al.: MIMIC-CXR-JPG, a large publicly available database of labeled chest radiographs. arXiv preprint arXiv:1901.07042 (2019)","DOI":"10.1038\/s41597-019-0322-0"},{"key":"36_CR36","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2022.102510","volume":"80","author":"S Yang","year":"2022","unstructured":"Yang, S., Wu, X., Ge, S., et al.: Knowledge matters: chest radiology report generation with general and specific knowledge. Med. Image Anal. 80, 102510 (2022)","journal-title":"Med. Image Anal."},{"issue":"3","key":"36_CR37","first-page":"100033","volume":"1","author":"W Zhanyu","year":"2023","unstructured":"Zhanyu, W., Lingqiao, L., Lei, W., Luping, Z.: R2GenGPT: radiology report generation with frozen LLMs. CoRR 1(3), 100033 (2023)","journal-title":"CoRR"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Chinese Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-9437-9_36","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T16:33:37Z","timestamp":1730392417000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-9437-9_36"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9789819794362","9789819794379"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-9437-9_36","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLPCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"CCF International Conference on Natural Language Processing and Chinese Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hangzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 November 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nlpcc2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tcci.ccf.org.cn\/conference\/2024\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}