{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T04:58:29Z","timestamp":1743137909422,"version":"3.40.3"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031781063"},{"type":"electronic","value":"9783031781070"}],"license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-78107-0_7","type":"book-chapter","created":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T19:31:38Z","timestamp":1733081498000},"page":"101-116","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Customizable and\u00a0Programmable Deep Learning"],"prefix":"10.1007","author":[{"given":"Ratnabali","family":"Pal","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Samarjit","family":"Kar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0706-2565","authenticated-orcid":false,"given":"Arif Ahmed","family":"Sekh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"key":"7_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2022.104424","volume":"81","author":"F Demir","year":"2023","unstructured":"Demir, F., Akbulut, Y., Ta\u015fc\u0131, B., Demir, K.: Improving brain tumor classification performance with an effective approach based on new deep learning model named 3ACL from 3D MRI data. Biomed. Signal Process. Control 81, 104424 (2023)","journal-title":"Biomed. Signal Process. Control"},{"issue":"5","key":"7_CR2","doi-asserted-by":"publisher","first-page":"1122","DOI":"10.1109\/JAS.2023.123618","volume":"10","author":"W Tianyu","year":"2023","unstructured":"Tianyu, W., et al.: A brief overview of ChatGPT: the history, status quo and potential future development. IEEE\/CAA J. Automatica Sinica 10(5), 1122\u20131136 (2023)","journal-title":"IEEE\/CAA J. Automatica Sinica"},{"key":"7_CR3","unstructured":"Team, G., et\u00a0al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"7_CR4","unstructured":"Marcus, G., Davis, E., Aaronson, S.: A very preliminary analysis of DALL-E 2. arXiv preprint arXiv:2204.13807 (2022)"},{"key":"7_CR5","doi-asserted-by":"crossref","unstructured":"Koonce, B., Koonce, B.: ResNet 50. Convolutional neural networks with swift for tensorflow: image recognition and dataset categorization, pp. 63\u201372 (2021)","DOI":"10.1007\/978-1-4842-6168-2_6"},{"key":"7_CR6","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1016\/j.neucom.2021.03.091","volume":"452","author":"Z Niu","year":"2021","unstructured":"Niu, Z., Zhong, G., Hui, Yu.: A review on the attention mechanism of deep learning. Neurocomputing 452, 48\u201362 (2021)","journal-title":"Neurocomputing"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Savci, P., Das, B.: Comparison of pre-trained language models in terms of carbon emissions, time and accuracy in multi-label text classification using AutoML. Heliyon 9(5), e15670 (2023)","DOI":"10.1016\/j.heliyon.2023.e15670"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Deng, A., Li, X., Hu, D., Wang, T., Xiong, H., Xu, C.-Z.: Towards inadequately pre-trained models in transfer learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19397\u201319408 (2023)","DOI":"10.1109\/ICCV51070.2023.01777"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Wang, H., Li, J., Wu, H., Hovy, E., Sun, Y.: Pre-trained language models and their applications. Engineering 25, 51\u201365 (2022)","DOI":"10.1016\/j.eng.2022.04.024"},{"key":"7_CR10","unstructured":"Boyko, J., et\u00a0al.: An interdisciplinary outlook on large language models for scientific research. arXiv preprint arXiv:2311.04929 (2023)"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Ooi, K.-B., et\u00a0al.: The potential of generative artificial intelligence across disciplines: perspectives and future directions. J. Comput. Inf. Syst. 1\u201332 (2023)","DOI":"10.1080\/08874417.2023.2261010"},{"key":"7_CR12","doi-asserted-by":"crossref","unstructured":"Le, D., Keren, G., Chan, J., Mahadeokar, J., Fuegen, C., Seltzer, M.L.: Deep shallow fusion for RNN-T personalization. In: 2021 IEEE Spoken Language Technology Workshop (SLT), pp. 251\u2013257. IEEE (2021)","DOI":"10.1109\/SLT48900.2021.9383560"},{"key":"7_CR13","doi-asserted-by":"crossref","unstructured":"Velasco, L., et al.: End-to-end intent-based networking. IEEE Commun. Mag. 59(10), 106\u2013112 (2021)","DOI":"10.1109\/MCOM.101.2100141"},{"key":"7_CR14","unstructured":"Liu, X., Chen, Y., Li, H., Li, B., Zhao, D.: Cross-domain random pre-training with prototypes for reinforcement learning. arXiv preprint arXiv:2302.05614 (2023)"},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Basiri, M.E., Nemati, S., Abdar, M., Asadi, S., Acharrya, U.R.: A novel fusion-based deep learning model for sentiment analysis of COVID-19 tweets. Knowl. Based Syst. 228, 107242 (2021)","DOI":"10.1016\/j.knosys.2021.107242"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Chakraborty, A., Joardar, S., Sekh, A.A.: Ensemble classifier for Hindi hostile content detection. ACM Trans. Asian Low-Resour. Lang. Inf. Process. 23(1), 1\u201317 (2024)","DOI":"10.1145\/3591353"},{"issue":"5","key":"7_CR17","doi-asserted-by":"publisher","first-page":"829","DOI":"10.1162\/neco_a_01273","volume":"32","author":"J Gao","year":"2020","unstructured":"Gao, J., Li, P., Chen, Z., Zhang, J.: A survey on deep learning for multimodal data fusion. Neural Comput. 32(5), 829\u2013864 (2020)","journal-title":"Neural Comput."},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Wang, R., et\u00a0al.: K-adapter: Infusing knowledge into pre-trained models with adapters. arXiv preprint arXiv:2002.01808 (2020)","DOI":"10.18653\/v1\/2021.findings-acl.121"},{"key":"7_CR19","unstructured":"Pantazis, O., Brostow, G., Jones, K., Aodha, O.M.: SVL-adapter: Self-supervised adapter for vision-language pretrained models. arXiv preprint arXiv:2210.03794 (2022)"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Thakare, K.V., Sharma, N., Dogra, D.P., Choi, H., Kim, I.-J.: A multi-stream deep neural network with late fuzzy fusion for real-world anomaly detection. Expert Syst. Appl. 201, 117030 (2022)","DOI":"10.1016\/j.eswa.2022.117030"},{"issue":"5","key":"7_CR21","doi-asserted-by":"publisher","first-page":"2189","DOI":"10.1109\/TIP.2018.2795742","volume":"27","author":"M Saha","year":"2018","unstructured":"Saha, M., Chakraborty, C.: Her2Net: a deep framework for semantic segmentation and classification of cell membranes and nuclei in breast cancer evaluation. IEEE Trans. Image Process. 27(5), 2189\u20132200 (2018)","journal-title":"IEEE Trans. Image Process."},{"key":"7_CR22","doi-asserted-by":"publisher","first-page":"3367","DOI":"10.1109\/TIP.2023.3276570","volume":"32","author":"H Li","year":"2023","unstructured":"Li, H., Huang, J., Jin, P., Song, G., Qi, W., Chen, J.: Weakly-supervised 3D spatial reasoning for text-based visual question answering. IEEE Trans. Image Process. 32, 3367\u20133382 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Yang, Z., et al.: TAP: text-aware pre-training for text-VQA and text-caption. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8751\u20138761 (2021)","DOI":"10.1109\/CVPR46437.2021.00864"},{"key":"7_CR24","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: VizWiz grand challenge: answering visual questions from blind people. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3608\u20133617 (2018)","DOI":"10.1109\/CVPR.2018.00380"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: VizWiz-Priv: a dataset for recognizing the presence and purpose of private visual information in images taken by blind people. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 939\u2013948 (2019)","DOI":"10.1109\/CVPR.2019.00103"},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Akula, A., Changpinyo, S., Gong, B., Sharma, P., Zhu, S.-C., Soricut, R.: CrossVQA: scalably generating benchmarks for systematically testing VQA generalization. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 2148\u20132166 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.164"},{"key":"7_CR27","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"7_CR28","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: VQA: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"7_CR29","doi-asserted-by":"crossref","unstructured":"Schwenk, D., Khandelwal, A., Clark, C., Marino, K., Mottaghi, R.: A-OKVQA: a benchmark for visual question answering using world knowledge. In: European Conference on Computer Vision, pp. 146\u2013162. Springer (2022)","DOI":"10.1007\/978-3-031-20074-8_9"},{"issue":"1","key":"7_CR30","doi-asserted-by":"publisher","first-page":"54","DOI":"10.1007\/s44196-023-00233-6","volume":"16","author":"L Siyu","year":"2023","unstructured":"Siyu, L., Ding, Y., Liu, M., Yin, Z., Yin, L., Zheng, W.: Multiscale feature extraction and fusion of image and text in VQA. Int. J. Comput. Intell. Syst. 16(1), 54 (2023)","journal-title":"Int. J. Comput. Intell. Syst."},{"key":"7_CR31","unstructured":"Jung, B., Gu, L., Harada, T.: bumjun_jung at VQA-Med 2020: VQA model based on feature extraction and multi-modal feature fusion. In: CLEF (Working Notes) (2020)"},{"key":"7_CR32","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108214","volume":"122","author":"W Jiajia","year":"2022","unstructured":"Jiajia, W., et al.: A multimodal attention fusion network with a dynamic vocabulary for textVQA. Pattern Recogn. 122, 108214 (2022)","journal-title":"Pattern Recogn."},{"key":"7_CR33","doi-asserted-by":"crossref","unstructured":"Wang, A., et al.: A novel deep learning-based 3D cell segmentation framework for future image-based disease detection. Sci. Rep. 12(1), 342 (2022)","DOI":"10.1038\/s41598-021-04048-3"},{"key":"7_CR34","doi-asserted-by":"crossref","unstructured":"Masoudi, S., et al.: Quick guide on radiology image pre-processing for deep learning applications in prostate cancer research. J. Med. Imaging 8(1), 010901\u2013010901 (2021)","DOI":"10.1117\/1.JMI.8.1.010901"},{"key":"7_CR35","volume":"115","author":"Yu Wenhao","year":"2022","unstructured":"Wenhao, Yu., Huang, Q.: A deep encoder-decoder network for anomaly detection in driving trajectory behavior under spatio-temporal context. Int. J. Appl. Earth Obs. Geoinf. 115, 103115 (2022)","journal-title":"Int. J. Appl. Earth Obs. Geoinf."},{"key":"7_CR36","doi-asserted-by":"crossref","unstructured":"Islam, S.M., Joardar, S., Sekh, A.A.: DSSN: dual shallow Siamese network for fashion image retrieval. Multimedia Tools Appl. 82(11), 16501\u201316517 (2023)","DOI":"10.1007\/s11042-022-14204-0"},{"key":"7_CR37","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et al.: Knowledgeable preference alignment for LLMs in domain-specific question answering. arXiv preprint arXiv:2311.06503 (2023)","DOI":"10.18653\/v1\/2024.findings-acl.52"},{"key":"7_CR38","unstructured":"Du, Y., et\u00a0al.: PP-OCR: A practical ultra lightweight OCR system. arXiv preprint arXiv:2009.09941 (2020)"},{"key":"7_CR39","unstructured":"Kim, W., Son, B., Kim, I.: ViLT: vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning, pp. 5583\u20135594. PMLR (2021)"},{"key":"7_CR40","unstructured":"Mokady, R., Hertz, A., Bermano, A.H.: ClipCap: Clip prefix for image captioning. arXiv preprint arXiv:2111.09734 (2021)"},{"key":"7_CR41","unstructured":"Zhang, J., Zhao, Y., Saleh, M., Liu, P.: PEGASUS: pre-training with extracted gap-sentences for abstractive summarization. In: International Conference on Machine Learning, pp. 11328\u201311339. PMLR (2020)"},{"key":"7_CR42","unstructured":"Sanh, V., Debut, L., Chaumond, J., Wolf, T.: DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108 (2019)"},{"key":"7_CR43","doi-asserted-by":"crossref","unstructured":"Song, H., Dong, L., Zhang, W.-N., Liu, T., Wei, F.: Clip models are few-shot learners: Empirical studies on VQA and visual entailment. arXiv preprint arXiv:2203.07190 (2022)","DOI":"10.18653\/v1\/2022.acl-long.421"},{"key":"7_CR44","doi-asserted-by":"crossref","unstructured":"Sung, Y.-L., Cho, J., Bansal, M.: VL-adapter: parameter-efficient transfer learning for vision-and-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5227\u20135237 (2022)","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"7_CR45","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: TrOCR: transformer-based optical character recognition with pre-trained models. In Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 13094\u201313102 (2023)","DOI":"10.1609\/aaai.v37i11.26538"},{"key":"7_CR46","doi-asserted-by":"crossref","unstructured":"Ullah, F., et al.: Brain MR image enhancement for tumor segmentation using 3D U-Net. Sensors 21(22), 7528 (2021)","DOI":"10.3390\/s21227528"},{"issue":"1","key":"7_CR47","doi-asserted-by":"publisher","first-page":"393","DOI":"10.1109\/TII.2019.2938527","volume":"16","author":"R Nawaratne","year":"2019","unstructured":"Nawaratne, R., Alahakoon, D., De Silva, D., Xinghuo, Yu.: Spatiotemporal anomaly detection using deep learning for real-time video surveillance. IEEE Trans. Industr. Inf. 16(1), 393\u2013402 (2019)","journal-title":"IEEE Trans. Industr. Inf."}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-78107-0_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T20:03:59Z","timestamp":1733083439000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-78107-0_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"ISBN":["9783031781063","9783031781070"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-78107-0_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}