{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:05:43Z","timestamp":1784217943504,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":41,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819203680","type":"print"},{"value":"9789819203697","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-92-0369-7_8","type":"book-chapter","created":{"date-parts":[[2026,5,11]],"date-time":"2026-05-11T11:27:04Z","timestamp":1778498824000},"page":"116-132","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Set-to-One Structured Captioning for\u00a0Heterogeneous Nasal Image Collections via\u00a0Spatio-Temporal Modeling and\u00a0Clinical Knowledge Integration"],"prefix":"10.1007","author":[{"given":"Xinpan","family":"Yuan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianuo","family":"Ju","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liujie","family":"Hua","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingzhu","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,12]]},"reference":[{"issue":"1","key":"8_CR1","doi-asserted-by":"publisher","first-page":"18619","DOI":"10.1038\/s41598-024-69827-0","volume":"14","author":"Y Rao","year":"2024","unstructured":"Rao, Y., et al.: Automated diagnosis of adenoid hypertrophy with lateral cephalogram in children based on multi-scale local attention. Sci. Rep. 14(1), 18619 (2024)","journal-title":"Sci. Rep."},{"key":"8_CR2","doi-asserted-by":"crossref","unstructured":"Yuan, X., Jin, S., Hua, L., Zhao, G., Zhang, C., Guo, Y.: YUANF-AR relation aware representation learning for lesion image segmentation and grading. In: ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE (2025)","DOI":"10.1109\/ICASSP49660.2025.10888488"},{"issue":"2","key":"8_CR3","doi-asserted-by":"publisher","first-page":"14","DOI":"10.3390\/ohbm5020014","volume":"5","author":"K Bayer","year":"2024","unstructured":"Bayer, K., et al.: Nasal septal deviation classifications associated with revision septoplasty. J. Otorhinolaryngol. Hear. Balance Med. 5(2), 14 (2024)","journal-title":"J. Otorhinolaryngol. Hear. Balance Med."},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Yuan, X., Kuang, J., Hua, L., Zhao, G., Zhang, C., Li, J.: A novel single continuous shot multiple lesions endoscopy report generation. In: ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE (2025)","DOI":"10.1109\/ICASSP49660.2025.10888634"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Phoommanee, N., Andrews, P.J., Leung, T.S.: Segmentation of endoscopy images of the anterior nasal cavity using deep learning. In: Medical Imaging 2024: Computer-Aided Diagnosis, vol. 12927, pp. 11\u201315. SPIE (2024)","DOI":"10.1117\/12.2691427"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Zhang, Z., et al.: Sam-guided enhanced fine-grained encoding with mixed semantic learning for medical image captioning. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1731\u20131735. IEEE (2024)","DOI":"10.1109\/ICASSP48485.2024.10446878"},{"key":"8_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2024.103377","volume":"99","author":"W Lang","year":"2025","unstructured":"Lang, W., Liu, Z., Zhang, Y.: DACG: dual attention and context guidance model for radiology report generation. Med. Image Anal. 99, 103377 (2025)","journal-title":"Med. Image Anal."},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Park, S.-J., Heo, K.S., Shin, D.H., Son, Y.H., Oh, J.H., Kam, T.E.: Dart: disease-aware image-text alignment and self-correcting re-alignment for trustworthy radiology report generation. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 15580\u201315589 (2025)","DOI":"10.1109\/CVPR52734.2025.01452"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Wang, X., et al.: Cxpmrg-bench: pre-training and benchmarking for x-ray medical report generation on chexpert plus dataset. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 5123\u20135133 (2025)","DOI":"10.1109\/CVPR52734.2025.00483"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Liu, H., et al.: Protecting your video content: disrupting automated video-based LLM annotations. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 24056\u201324065 (2025)","DOI":"10.1109\/CVPR52734.2025.02240"},{"issue":"11","key":"8_CR11","doi-asserted-by":"publisher","first-page":"13619","DOI":"10.1007\/s10462-023-10488-2","volume":"56","author":"H Sharma","year":"2023","unstructured":"Sharma, H., Padha, D.: A comprehensive survey on image captioning: from handcrafted to deep learning-based techniques, a taxonomy and open research issues. Artif. Intell. Rev. 56(11), 13619\u201313661 (2023)","journal-title":"Artif. Intell. Rev."},{"issue":"7","key":"8_CR12","doi-asserted-by":"publisher","first-page":"2657","DOI":"10.1109\/TMI.2024.3372638","volume":"43","author":"A Liu","year":"2024","unstructured":"Liu, A., Guo, Y., Yong, J., Feng, X.: Multi-grained radiology report generation with sentence-level image-language contrastive learning. IEEE Trans. Med. Imaging 43(7), 2657\u20132669 (2024)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"8_CR13","unstructured":"Ram, S., Vinoth, S., Gopalakrishnan, R.N., Balakumar, A.A., Kalinathan, L., Velankanni, T.A.J., Leveraging diverse CNN architectures for medical image captioning: Densenet-121, mobilenetv2, and resnet-50 in imageclef,: In: CLEF2024 Working Notes, CEUR Workshop Proceedings, p. 2024. CEUR-WS. org, Grenoble (2024)"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Raphael, A.,\u00a0Abisri, S.,\u00a0Anitha, E.,\u00a0Ritika, S., Venugopalan, M.: Attention based cnn-rnn hybrid model for image captioning. In: 2024 5th IEEE Global Conference for Advancement in Technology (GCAT), pp. 1\u20135. IEEE (2024)","DOI":"10.1109\/GCAT62922.2024.10923871"},{"issue":"3","key":"8_CR15","doi-asserted-by":"publisher","DOI":"10.1016\/j.metrad.2023.100033","volume":"1","author":"Z Wang","year":"2023","unstructured":"Wang, Z., Liu, L., Wang, L., Zhou, L.: R2gengpt: radiology report generation with frozen LLMs. Meta-Radiology 1(3), 100033 (2023)","journal-title":"Meta-Radiology"},{"key":"8_CR16","doi-asserted-by":"publisher","unstructured":"Sha, Y., Pan, H., Meng, W., Li, K.: Contrastive knowledge-guided large language models for medical report generation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 111\u2013120. Springer, Heidelberg (2025). https:\/\/doi.org\/10.1007\/978-3-032-04978-0_11","DOI":"10.1007\/978-3-032-04978-0_11"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"Yun, H., Maeng, J., Kang, E., Suk, H.I.: DIFF-RRG: longitudinal disease-wise patch difference as guidance for LLM-based radiology report generation. In:International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 152\u2013161. Springer, Heidelberg (2025)","DOI":"10.1007\/978-3-032-04981-0_15"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Heiman, A., Zhang, X., Chen, E., Kim, S.E., Rajpurkar, P.: Factchexcker: mitigating measurement hallucinations in chest x-ray report generation models. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 30787\u201330796 (2025)","DOI":"10.1109\/CVPR52734.2025.02867"},{"key":"8_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110744","volume":"156","author":"M Tian","year":"2024","unstructured":"Tian, M., Li, G., Qi, Y., Wang, S., Sheng, Q.Z., Huang, Q.: Rethink video retrieval representation for video captioning. Pattern Recogn. 156, 110744 (2024)","journal-title":"Pattern Recogn."},{"key":"8_CR20","doi-asserted-by":"crossref","unstructured":"Huang, B., Wang, X., Chen, H., Song, Z., Zhu, W.: Vtimellm: Empower LLM to grasp video moments. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14271\u201314280 (2024)","DOI":"10.1109\/CVPR52733.2024.01353"},{"key":"8_CR21","unstructured":"Ma, D., et\u00a0al.: Iv-bench: a benchmark for image-grounded video perception and reasoning in multimodal LLMs. arXiv preprint arXiv:2504.15415 (2025)"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Niu, J., et\u00a0al.: Ovo-bench: how far is your video-LLMs from real-world online video understanding? In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 18902\u201318913 (2025)","DOI":"10.1109\/CVPR52734.2025.01761"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Li, H., Wang, H., Sun, X., He, H., Feng, J.: Prompt-guided generation of structured chest x-ray report using a pre-trained LLM. In: 2024 IEEE International Conference on Multimedia and Expo (ICME), pp. 1\u20136. IEEE (2024)","DOI":"10.1109\/ICME57554.2024.10687707"},{"issue":"1","key":"8_CR24","first-page":"21","volume":"2","author":"X Wang","year":"2021","unstructured":"Wang, X., Zhang, Y., Guo, Z., Li, J.: TMRGM: a template-based multi-attention model for x-ray imaging report generation. J. Artif. Intell. Med. Sci. 2(1), 21\u201332 (2021)","journal-title":"J. Artif. Intell. Med. Sci."},{"key":"8_CR25","doi-asserted-by":"crossref","unstructured":"Javed, U., Abbas, T., Raza, M., Mehmood, F., Yaqoob, J., Li, H.: An improved medical visual question answering model based on clip and bert. In: 2024 International Conference on Image Processing, Computer Vision and Machine Learning (ICICML), pp. 743\u2013748. IEEE (2024)","DOI":"10.1109\/ICICML63543.2024.10958067"},{"key":"8_CR26","unstructured":"Karimian, B., Avanzato, G., Belharbi, S., McCaffrey, L., Shateri, M., Granger, E.: Clip-it: clip-based pairing for histology images classification. arXiv preprint arXiv:2504.16181 (2025)"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Jin, H., Che, H., Lin, Y., Chen, H.: Promptmrg: diagnosis-driven prompts for medical report generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 2607\u20132615 (2024)","DOI":"10.1609\/aaai.v38i3.28038"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Chen, Z., Song, Y., Chang, T.H., Wan, X.: Generating radiology reports via memory-driven transformer. arXiv preprint arXiv:2010.16056 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.112"},{"key":"8_CR29","unstructured":"Chen, Z., Shen, Y., Song, Y., Wan, X.: Cross-modal memory networks for radiology report generation. arXiv preprint arXiv:2204.13258 (2022)"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Liu, T., et al.: HC-LLM: historical-constrained large language models for radiology report generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 5595\u20135603 (2025)","DOI":"10.1609\/aaai.v39i6.32596"},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Pavlopoulos, J., Kougia, V., Androutsopoulos, I.: A survey on biomedical image captioning. In: Proceedings of the Second Workshop on Shortcomings in Vision and Language, pp. 26\u201336 (2019)","DOI":"10.18653\/v1\/W19-1803"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Johnson, A.E.W., et al.: MIMIC-CXR-JPG, a large publicly available database of labeled chest radiographs. arXiv preprint arXiv:1901.07042 (2019)","DOI":"10.1038\/s41597-019-0322-0"},{"key":"8_CR33","doi-asserted-by":"crossref","unstructured":"Post, M.: A call for clarity in reporting bleu scores. arXiv preprint arXiv:1804.08771 (2018)","DOI":"10.18653\/v1\/W18-6319"},{"key":"8_CR34","doi-asserted-by":"crossref","unstructured":"Graham, Y.: Re-evaluating automatic summarization with bleu and 192 shades of rouge. In: Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing, pp. 128\u2013137 (2015)","DOI":"10.18653\/v1\/D15-1013"},{"key":"8_CR35","unstructured":"Banerjee, S., Lavie, A.: Meteor: an automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp. 65\u201372 (2005)"},{"key":"8_CR36","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Zitnick, C.L., Parikh, D.: Cider: consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4566\u20134575 (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"8_CR37","doi-asserted-by":"publisher","unstructured":"Naidu, G., Zuva, T., Sibanda, E.M.: A review of evaluation metrics in machine learning algorithms. In: Computer Science On-line Conference, pp. 15\u201325. Springer, Heidelberg (2023). https:\/\/doi.org\/10.1007\/978-3-031-35314-7_2","DOI":"10.1007\/978-3-031-35314-7_2"},{"key":"8_CR38","doi-asserted-by":"crossref","unstructured":"Irvin, J., et al.: Chexpert: a large chest radiograph dataset with uncertainty labels and expert comparison. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 590\u2013597 (2019)","DOI":"10.1609\/aaai.v33i01.3301590"},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Wang, Z., Sun, Y., Li, Z., Yang, X., Chen, F., Liao, H.: LLM-rg4: flexible and factual radiology report generation across diverse input contexts. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 8250\u20138258 (2025)","DOI":"10.1609\/aaai.v39i8.32890"},{"key":"8_CR40","doi-asserted-by":"crossref","unstructured":"Liu, C., Tian, Y., Chen, W., Song, Y., Zhang, Y.: Bootstrapping large language models for radiology report generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 18635\u201318643 (2024)","DOI":"10.1609\/aaai.v38i17.29826"},{"issue":"3","key":"8_CR41","doi-asserted-by":"publisher","first-page":"943","DOI":"10.1038\/s41591-024-03423-7","volume":"31","author":"K Singhal","year":"2025","unstructured":"Singhal, K., et al.: Toward expert-level medical question answering with large language models. Nat. Med. 31(3), 943\u2013950 (2025)","journal-title":"Nat. Med."}],"container-title":["Lecture Notes in Computer Science","Database Systems for Advanced Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-0369-7_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T15:23:49Z","timestamp":1784215429000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-0369-7_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819203680","9789819203697"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-0369-7_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"12 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DASFAA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Database Systems for Advanced Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Jeju","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 April 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dasfaa2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/dasfaa2026.github.io\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}