{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:45:39Z","timestamp":1782949539191,"version":"3.54.5"},"publisher-location":"Cham","reference-count":22,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031936968","type":"print"},{"value":"9783031936975","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T00:00:00Z","timestamp":1753315200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-93697-5_7","type":"book-chapter","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T13:48:45Z","timestamp":1753278525000},"page":"87-101","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Optimizing Image Captioning Using BLIP Framework with Advanced Processing of DWT LL Band-Compressed Images"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3104-0497","authenticated-orcid":false,"given":"M.","family":"Nivedita","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5030-1165","authenticated-orcid":false,"given":"Y.","family":"Asnath Victy Phamila","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9746-2962","authenticated-orcid":false,"given":"Krishna Prasad Y. V. S.","family":"Puram","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joel","family":"Fredrick","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,24]]},"reference":[{"key":"7_CR1","doi-asserted-by":"crossref","unstructured":"Li, B., Giannakis, G.: Enhancing sharpness-aware optimization through variance suppression. Adv. Neural Inform. Process. Syst. 36 (2024)","DOI":"10.52202\/075280-3104"},{"issue":"2","key":"7_CR2","doi-asserted-by":"publisher","first-page":"336","DOI":"10.1007\/s11390-024-3414-z","volume":"39","author":"ZN Li","year":"2024","unstructured":"Li, Z.N., Chen, X.H., Guo, S.N., Wang, S.Q., Pun, C.M.: Wavenhancer: unifying wavelet and transformer for image enhancement. J. Comput. Sci. Technol. 39(2), 336\u2013345 (2024)","journal-title":"J. Comput. Sci. Technol."},{"issue":"3","key":"7_CR3","doi-asserted-by":"publisher","first-page":"001","DOI":"10.30574\/gjeta.2024.19.3.0091","volume":"19","author":"SM Saadi","year":"2024","unstructured":"Saadi, S.M., Al-Jawher, W.: Enhancing image authenticity: a new approach for binary fake image classification using DWT and swin transformer. Global J. Eng. Technol. Adv. 19(3), 001\u2013010 (2024)","journal-title":"Global J. Eng. Technol. Adv."},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Meng, G., Huang, J., Wang, Y., Fu, Z., Ding, X., Huang, Y.: Progressive high-frequency reconstruction for pan-sharpening with implicit neural representation. In:\u00a0Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 5, pp. 4189\u20134197 (2024)","DOI":"10.1609\/aaai.v38i5.28214"},{"issue":"8","key":"7_CR5","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1007\/s42452-024-06111-w","volume":"6","author":"N Shukla","year":"2024","unstructured":"Shukla, N., Sood, M., Kumar, A., Choudhary, G.: Adaptive decomposition with guided filtering and Laplacian pyramid-based image fusion method for medical applications. Discover Appl. Sci. 6(8), 423 (2024)","journal-title":"Discover Appl. Sci."},{"key":"7_CR6","first-page":"255","volume":"1","author":"V Vysotska","year":"2024","unstructured":"Vysotska, V., Sharonova, N., Shirokopetleva, M., Dolhanenko, O., Chupryna, A., Smelyakov, S.: Research of methods for image sharpness evaluation in photos of people. COLINS 1, 255\u2013272 (2024)","journal-title":"COLINS"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Pal, M., Biswas, T., Basuli, K., Biswas, B.: A novel parallel mammogram sharpening framework using modified Laplacian filter for lumps identification on GPU.\u00a0Innov. Syst. Softw. Eng. 1\u201317 (2024)","DOI":"10.21203\/rs.3.rs-3211944\/v1"},{"issue":"5","key":"7_CR8","doi-asserted-by":"publisher","first-page":"14363","DOI":"10.1007\/s11042-023-15799-8","volume":"83","author":"S Roy","year":"2024","unstructured":"Roy, S., Bhalla, K., Patel, R.: Mathematical analysis of histogram equalization techniques for medical image enhancement: a tutorial from the perspective of data loss. Multimedia Tools Appl. 83(5), 14363\u201314392 (2024)","journal-title":"Multimedia Tools Appl."},{"issue":"3","key":"7_CR9","first-page":"1","volume":"8","author":"R Radhika","year":"2024","unstructured":"Radhika, R., Mahajan, R.: A novel framework of hybrid optimization techniques for contrast-enhancement in cardiac MRI medical images. J. Angiotherapy 8(3), 1\u201311 (2024)","journal-title":"J. Angiotherapy"},{"issue":"03","key":"7_CR10","doi-asserted-by":"publisher","first-page":"24","DOI":"10.37547\/ajast\/Volume04Issue03-05","volume":"4","author":"ES Ummatovich","year":"2024","unstructured":"Ummatovich, E.S.: Data filtering in the image processing toolbox (IPT) environment. Am. J Appl. Sci. Technol. 4(03), 24\u201328 (2024)","journal-title":"Am. J Appl. Sci. Technol."},{"key":"7_CR11","unstructured":"Waghamare, M.P.U.: Enhanced analysis on image enrichment methods.\u00a0J. Public. Int. Res. Eng. Manage.\u00a010(07) (2024)"},{"issue":"1","key":"7_CR12","doi-asserted-by":"publisher","DOI":"10.2196\/56627","volume":"12","author":"U Naseem","year":"2024","unstructured":"Naseem, U., Thapa, S., Masood, A.: Advancing accuracy in multimodal medical tasks through bootstrapped language-image pretraining (BioMedBLIP): performance evaluation study. JMIR Med. Inform. 12(1), e56627 (2024)","journal-title":"JMIR Med. Inform."},{"key":"7_CR13","doi-asserted-by":"crossref","unstructured":"Lu, J., et al.: Unified-IO 2: scaling autoregressive multimodal models with vision language audio and action. In:\u00a0Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26439\u201326455 (2024)","DOI":"10.1109\/CVPR52733.2024.02497"},{"key":"7_CR14","unstructured":"Song, P., et al.: Efficiently gluing pre-trained language and vision models for image captioning.\u00a0ACM Trans. Intell. Syst. Technol."},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Gao, Y., et al.: Enhancing vision-language pre-training with rich supervisions. In:\u00a0Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13480\u201313491 (2024)","DOI":"10.1109\/CVPR52733.2024.01280"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Sheng, D., et al.: Towards more unified in-context visual understanding. In:\u00a0Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13362\u201313372 (2024)","DOI":"10.1109\/CVPR52733.2024.01269"},{"key":"7_CR17","doi-asserted-by":"crossref","unstructured":"Jin, P., Takanobu, R., Zhang, W., Cao, X., Yuan, L.: Chat-univi: Unified visual representation empowers large language models with image and video understanding. In:\u00a0Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13700\u201313710 (2024)","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Yue, Z., Shi, M., Wang, H., Ding, S., Chen, Q., Yang, S.: Bootstrapping vision-language models for self-supervised remote physiological measurement.\u00a0arXiv preprint arXiv:2407.08507 (2024)","DOI":"10.1007\/s11263-025-02388-5"},{"key":"7_CR19","doi-asserted-by":"crossref","unstructured":"Yang, C., Li, Z., Zhang, L.: Bootstrapping interactive image-text alignment for remote sensing image captioning.\u00a0IEEE Trans. Geosci. Remote Sens. (2024)","DOI":"10.1109\/TGRS.2024.3359316"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Li, R., Wu, Y., He, X.: Learning by correction: efficient tuning task for zero-shot generative vision-language reasoning. In:\u00a0Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13428\u201313437 (2024)","DOI":"10.1109\/CVPR52733.2024.01275"},{"key":"7_CR21","unstructured":"Zhang, H., Pustokhina, I., Pankratov, A.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv (2023)"},{"key":"7_CR22","unstructured":"Wang, J., et al.:. GIT: a generative image-to-text transformer for vision and language. arXiv (2023)"}],"container-title":["Communications in Computer and Information Science","Computer Vision and Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-93697-5_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:21:52Z","timestamp":1782948112000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-93697-5_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,24]]},"ISBN":["9783031936968","9783031936975"],"references-count":22,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-93697-5_7","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,24]]},"assertion":[{"value":"24 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CVIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computer Vision and Image Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chennai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cvip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/cvip2024.iiitdm.ac.in\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}