{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T20:24:44Z","timestamp":1774124684467,"version":"3.50.1"},"reference-count":78,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2024,5,17]],"date-time":"2024-05-17T00:00:00Z","timestamp":1715904000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,17]],"date-time":"2024-05-17T00:00:00Z","timestamp":1715904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Key R&D Program of Planned Science and Technology Project of Sichuan Province","award":["2021YFQ0054"],"award-info":[{"award-number":["2021YFQ0054"]}]},{"name":"SCITLAB","award":["SCITLAB-20004"],"award-info":[{"award-number":["SCITLAB-20004"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1007\/s13042-024-02177-5","type":"journal-article","created":{"date-parts":[[2024,5,17]],"date-time":"2024-05-17T12:02:08Z","timestamp":1715947328000},"page":"4617-4637","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Unified deep learning model for multitask representation and transfer learning: image classification, object detection, and image captioning"],"prefix":"10.1007","volume":"15","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9939-9821","authenticated-orcid":false,"given":"Leta Yobsan","family":"Bayisa","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weidong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingxian","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chiagoziem C.","family":"Ukwuoma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hirpesa Kebede","family":"Gutema","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ahmed","family":"Endris","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Turi","family":"Abu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,5,17]]},"reference":[{"key":"2177_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1080\/17460441.2021.1901685","volume":"16","author":"DZ Huang","year":"2021","unstructured":"Huang DZ, Baber JC, Bahmanyar SS (2021) The challenges of generalizability in artificial intelligence for ADME\/Tox endpoint and activity prediction. Expert Opin Drug Discov 16:1","journal-title":"Expert Opin Drug Discov"},{"key":"2177_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.micpro.2020.103051","author":"Q Fu","year":"2020","unstructured":"Fu Q, Wang C, Han X (2020) A CNN-LSTM network with attention approach for learning universal sentence representation in embedded system. Microprocess Microsyst. https:\/\/doi.org\/10.1016\/j.micpro.2020.103051","journal-title":"Microprocess Microsyst"},{"key":"2177_CR3","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2905617","author":"C Pang","year":"2019","unstructured":"Pang C, Liu H, Li X (2019) Multitask learning of time\u2013frequency CNN for sound source localization. IEEE Access. https:\/\/doi.org\/10.1109\/ACCESS.2019.2905617","journal-title":"IEEE Access"},{"key":"2177_CR4","doi-asserted-by":"crossref","unstructured":"Toshniwal S, Tang H, Lu L, Livescu K (2017) Multitask learning with low-level auxiliary tasks for encoder\u2013decoder based speech recognition. In: Proceedings of the annual conference of the international speech communication association, INTERSPEECH","DOI":"10.21437\/Interspeech.2017-1118"},{"key":"2177_CR5","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2019.2942548","author":"S Guo","year":"2020","unstructured":"Guo S, Zhang B, Yang T et al (2020) Multitask convolutional neural network with information fusion for bearing fault diagnosis and localization. IEEE Trans Ind Electron. https:\/\/doi.org\/10.1109\/TIE.2019.2942548","journal-title":"IEEE Trans Ind Electron"},{"key":"2177_CR6","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3061479","author":"G Kapidis","year":"2021","unstructured":"Kapidis G, Poppe R, Veltkamp RC (2021) Multi-dataset, multitask learning of egocentric vision tasks. IEEE Trans Pattern Anal Mach Intell. https:\/\/doi.org\/10.1109\/TPAMI.2021.3061479","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2177_CR7","doi-asserted-by":"crossref","unstructured":"Wang XE, Jain V, Ie E et al (2020) environment-agnostic multitask learning for natural language grounded navigation. In: Lecture notes in computer science (including subseries lecture notes in artificial intelligence and lecture notes in bioinformatics)","DOI":"10.1007\/978-3-030-58586-0_25"},{"key":"2177_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.10.013","author":"J Gu","year":"2018","unstructured":"Gu J, Wang Z, Kuen J et al (2018) Recent advances in convolutional neural networks. Pattern Recognit. https:\/\/doi.org\/10.1016\/j.patcog.2017.10.013","journal-title":"Pattern Recognit"},{"key":"2177_CR9","first-page":"1","volume":"16","author":"E Ben-Baruch","year":"2019","unstructured":"Ben-Baruch E, Ridnik T, Zamir N et al (2019) Attention Is All You Need. Adv Neural Inf Process Syst 16:1","journal-title":"Adv Neural Inf Process Syst"},{"key":"2177_CR10","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et al (2021) An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR 2021\u20149th international conference on learning representations"},{"key":"2177_CR11","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-021-04223-6","author":"LG Wright","year":"2022","unstructured":"Wright LG, Onodera T, Stein MM et al (2022) Deep physical neural networks trained with backpropagation. Nature. https:\/\/doi.org\/10.1038\/s41586-021-04223-6","journal-title":"Nature"},{"key":"2177_CR12","doi-asserted-by":"crossref","unstructured":"Strezoski G, Noord N, Worring M (2019) Many task learning with task routing. In: Proceedings of the IEEE international conference on computer vision","DOI":"10.1109\/ICCV.2019.00146"},{"key":"2177_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1093\/nsr\/nwx105","volume":"5","author":"Y Zhang","year":"2018","unstructured":"Zhang Y, Yang Q (2018) An overview of multi-task learning. Natl Sci Rev 5:1","journal-title":"Natl Sci Rev"},{"key":"2177_CR14","doi-asserted-by":"publisher","DOI":"10.1017\/ATSIP.2020.7","author":"Y Furusho","year":"2020","unstructured":"Furusho Y, Ikeda K (2020) Theoretical analysis of skip connections and batch normalization from generalization and optimization perspectives. APSIPA Trans Signal Inf Process. https:\/\/doi.org\/10.1017\/ATSIP.2020.7","journal-title":"APSIPA Trans Signal Inf Process"},{"key":"2177_CR15","first-page":"1","volume":"8","author":"N Jain","year":"2019","unstructured":"Jain N, Singh H, Sharma V (2019) Competitor analysis and benchmarking of improved Alex net. Int J Sci Technol Res 8:1","journal-title":"Int J Sci Technol Res"},{"key":"2177_CR16","first-page":"1","volume":"2020","author":"H Rampersad","year":"2020","unstructured":"Rampersad H (2020) FAST-RCNN. Total Perform Scorec 2020:1","journal-title":"Total Perform Scorec"},{"key":"2177_CR17","doi-asserted-by":"publisher","DOI":"10.3390\/app9081665","author":"Y Zhang","year":"2019","unstructured":"Zhang Y, Li D, Wang Y et al (2019) Abstract text summarization with a convolutional seq2seq model. Appl Sci. https:\/\/doi.org\/10.3390\/app9081665","journal-title":"Appl Sci"},{"key":"2177_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2021.103298","author":"S Warrier","year":"2022","unstructured":"Warrier S, Rutter EM, Flores KB (2022) Multitask neural networks for predicting bladder pressure with time series data. Biomed Signal Process Control. https:\/\/doi.org\/10.1016\/j.bspc.2021.103298","journal-title":"Biomed Signal Process Control"},{"key":"2177_CR19","doi-asserted-by":"crossref","unstructured":"Chou SH, Chao WL, Lai WS et al (2020) Visual question answering on 360\u00b0 images. In: Proceedings\u20142020 IEEE winter conference on applications of computer vision, WACV 2020","DOI":"10.1109\/WACV45572.2020.9093452"},{"key":"2177_CR20","doi-asserted-by":"publisher","DOI":"10.9781\/ijimai.2018.08.004","author":"S Jha","year":"2019","unstructured":"Jha S, Dey A, Kumar R, Kumar-Solanki V (2019) A novel approach on visual question answering by parameter prediction using faster region based convolutional neural network. Int J Interact Multimed Artif Intell. https:\/\/doi.org\/10.9781\/ijimai.2018.08.004","journal-title":"Int J Interact Multimed Artif Intell"},{"key":"2177_CR21","doi-asserted-by":"crossref","unstructured":"Xie J, Cai Y, Huang Q, Wang T (2021) Multiple objects-aware visual question generation. In: MM 2021\u2014proceedings of the 29th ACM international conference on multimedia","DOI":"10.1145\/3474085.3476969"},{"key":"2177_CR22","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput. https:\/\/doi.org\/10.1162\/neco.1997.9.8.1735","journal-title":"Neural Comput"},{"key":"2177_CR23","doi-asserted-by":"crossref","unstructured":"Toshevska M, Stojanovska F, Zdravevski E et al (2020) Exploration into deep learning text generation architectures for dense image captioning. In: Proceedings of the 2020 federated conference on computer science and information systems, FedCSIS 2020. pp 129\u2013136","DOI":"10.15439\/2020F57"},{"key":"2177_CR24","first-page":"1","volume":"2016","author":"A Rangamani","year":"2016","unstructured":"Rangamani A, Xiong T, Nair A et al (2016) Landmark detection and tracking in ultrasound using a CNN-RNN framework. Conf Neural Inf Process Syst 2016:1","journal-title":"Conf Neural Inf Process Syst"},{"key":"2177_CR25","doi-asserted-by":"crossref","unstructured":"Sun G, Probst T, Paudel DP et al (2021) Task switching network for multi-task learning. In: Proceedings of the IEEE international conference on computer vision","DOI":"10.1109\/ICCV48922.2021.00818"},{"key":"2177_CR26","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2947482","author":"J Yu","year":"2020","unstructured":"Yu J, Li J, Yu Z, Huang Q (2020) Multimodal transformer with multi-view visual representation for image captioning. IEEE Trans Circuits Syst Video Technol. https:\/\/doi.org\/10.1109\/TCSVT.2019.2947482","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"2177_CR27","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2987101","author":"Y Lan","year":"2020","unstructured":"Lan Y, Hao Y, Xia K et al (2020) Stacked residual recurrent neural networks with cross-layer attention for text classification. IEEE Access. https:\/\/doi.org\/10.1109\/ACCESS.2020.2987101","journal-title":"IEEE Access"},{"key":"2177_CR28","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.2985716","author":"Z Ji","year":"2022","unstructured":"Ji Z, Wang H, Han J, Pang Y (2022) SMAN: stacked multimodal attention network for cross-modal image-text retrieval. IEEE Trans Cybern. https:\/\/doi.org\/10.1109\/TCYB.2020.2985716","journal-title":"IEEE Trans Cybern"},{"key":"2177_CR29","unstructured":"Mittal S, Lamb A, Goyal A et al (2020) Learning to combine top-down and bottom-up signals in recurrent neural networks with attention over modules. In: 37th international conference on machine learning, ICML 2020"},{"key":"2177_CR30","volume-title":"Closing the loop between language and vision for embodied agents","author":"X Wang","year":"2020","unstructured":"Wang X, Wang WY, Wang Y-F (2020) Closing the loop between language and vision for embodied agents. University of California, Santa Barbara"},{"key":"2177_CR31","doi-asserted-by":"publisher","DOI":"10.3390\/s21092911","author":"J Lee","year":"2021","unstructured":"Lee J, Kim I (2021) Vision\u2013language\u2013knowledge co-embedding for visual commonsense reasoning. Sensors. https:\/\/doi.org\/10.3390\/s21092911","journal-title":"Sensors"},{"key":"2177_CR32","doi-asserted-by":"publisher","DOI":"10.25073\/2588-1086\/vnucsce.295","author":"N Van Tu","year":"2021","unstructured":"Van Tu N, Cuong LA (2021) A deep learning model of multiple knowledge sources integration for community question answering. VNU J Sci Comput Sci Commun Eng. https:\/\/doi.org\/10.25073\/2588-1086\/vnucsce.295","journal-title":"VNU J Sci Comput Sci Commun Eng"},{"key":"2177_CR33","doi-asserted-by":"publisher","unstructured":"Wang Y, Zhu M, Xu C et al (2022) Exploiting image captions and external knowledge as representation enhancement for VQA. Qinghua Daxue Xuebao\/J Tsinghua Univ. https:\/\/doi.org\/10.16511\/j.cnki.qhdxxb.2022.21.010","DOI":"10.16511\/j.cnki.qhdxxb.2022.21.010"},{"key":"2177_CR34","doi-asserted-by":"crossref","unstructured":"Wu J, Hu Z, Mooney RJ (2020) Generating question relevant captions to aid visual question answering. In: ACL 2019\u201457th annual meeting of the association for computational linguistics, proceedings of the conference","DOI":"10.18653\/v1\/P19-1348"},{"key":"2177_CR35","doi-asserted-by":"crossref","unstructured":"Lin P, Yang M (2020) A shared-private representation model with coarse-to-fine extraction for target sentiment analysis. In: Findings of the association for computational linguistics findings of ACL: EMNLP 2020","DOI":"10.18653\/v1\/2020.findings-emnlp.382"},{"key":"2177_CR36","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2022.3179011","author":"G Niu","year":"2023","unstructured":"Niu G, Liu E, Wang X et al (2023) Enhanced discriminate feature learning deep residual CNN for multitask bearing fault diagnosis with information fusion. IEEE Trans Ind Inform. https:\/\/doi.org\/10.1109\/TII.2022.3179011","journal-title":"IEEE Trans Ind Inform"},{"key":"2177_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.physa.2021.126744","author":"Y Liu","year":"2022","unstructured":"Liu Y, Li K, Yan D, Gu S (2022) A network-based CNN model to identify the hidden information in text data. Phys A Stat Mech its Appl. https:\/\/doi.org\/10.1016\/j.physa.2021.126744","journal-title":"Phys A Stat Mech its Appl"},{"key":"2177_CR38","doi-asserted-by":"crossref","unstructured":"Chen L, Zhang H, Xiao J et al (2017) SCA-CNN: spatial and channel-wise attention in convolutional networks for image captioning. In: Proceedings-30th IEEE conference on computer vision and pattern recognition, CVPR 2017","DOI":"10.1109\/CVPR.2017.667"},{"key":"2177_CR39","doi-asserted-by":"crossref","unstructured":"Graham B, El-Nouby A, Touvron H et al (2021) LeViT: a vision transformer in ConvNet\u2019s clothing for faster inference. In: Proceedings of the IEEE international conference on computer vision","DOI":"10.1109\/ICCV48922.2021.01204"},{"key":"2177_CR40","unstructured":"Tolstikhin I, Houlsby N, Kolesnikov A et al (2021) MLP-mixer: an all-MLP architecture for vision. In: Advances in neural information processing systems"},{"key":"2177_CR41","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2022.3185179","author":"HMD Kabir","year":"2022","unstructured":"Kabir HMD, Abdar M, Khosravi A et al (2022) SpinalNet: deep neural network with gradual input. IEEE Trans Artif Intell. https:\/\/doi.org\/10.1109\/TAI.2022.3185179","journal-title":"IEEE Trans Artif Intell"},{"key":"2177_CR42","doi-asserted-by":"crossref","unstructured":"Sudowe P, Leibe B (2016) Patchit: Self-supervised network weight initialization for fine-grained recognition. In: British machine vision conference 2016, BMVC 2016","DOI":"10.5244\/C.30.75"},{"key":"2177_CR43","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11038-0","author":"J Zakraoui","year":"2021","unstructured":"Zakraoui J, Saleh M, Al-Maadeed S, Jaam JM (2021) Improving text-to-image generation with object layout guidance. Multimed Tools Appl. https:\/\/doi.org\/10.1007\/s11042-021-11038-0","journal-title":"Multimed Tools Appl"},{"key":"2177_CR44","doi-asserted-by":"publisher","DOI":"10.1145\/3505244","author":"S Khan","year":"2022","unstructured":"Khan S, Naseer M, Hayat M et al (2022) Transformers in vision: a survey. ACM Comput Surv. https:\/\/doi.org\/10.1145\/3505244","journal-title":"ACM Comput Surv"},{"key":"2177_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1080\/01431161.2021.1954261","volume":"42","author":"GA Fotso Kamga","year":"2021","unstructured":"Fotso Kamga GA, Bitjoka L, Akram T et al (2021) Advancements in satellite image classification: methodologies, techniques, approaches and applications. Int J Remote Sens 42:1","journal-title":"Int J Remote Sens"},{"key":"2177_CR46","first-page":"1","volume":"2018","author":"ME Peters","year":"2018","unstructured":"Peters ME, Neumann M, Iyyer M et al (2018) Improving language understanding with unsupervised learning. OpenAI 2018:1","journal-title":"OpenAI"},{"key":"2177_CR47","doi-asserted-by":"publisher","DOI":"10.1007\/s12559-021-09828-7","author":"S Ghosh","year":"2022","unstructured":"Ghosh S, Ekbal A, Bhattacharyya P (2022) A multitask framework to detect depression, sentiment and multi-label emotion from suicide notes. Cognit Comput. https:\/\/doi.org\/10.1007\/s12559-021-09828-7","journal-title":"Cognit Comput"},{"key":"2177_CR48","unstructured":"Tan M, Le QV (2019) EfficientNet: rethinking model scaling for convolutional neural networks. In: 36th international conference on machine learning, ICML 2019"},{"key":"2177_CR49","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2016.90"},{"key":"2177_CR50","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger KQ (2017) DenseNet. In: Proceedings of the 30th IEEE conf comput vis pattern recognition, CVPR 2017 2017-Janua"},{"key":"2177_CR51","unstructured":"Ivan V, Slater D, Spacagna G, et al (2019) Python deep deep learning"},{"key":"2177_CR52","doi-asserted-by":"crossref","unstructured":"Sun Z, Sarma PK, Liang Y, Sethares WA (2021) A new view of multi-modal language analysis: Audio and video features as text \u201cStyles\u201d. In: EACL 2021-16th conference of the European chapter of the association for computational linguistics, proceedings of the conference","DOI":"10.18653\/v1\/2021.eacl-main.167"},{"key":"2177_CR53","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-020-19266-y","author":"IV Tetko","year":"2020","unstructured":"Tetko IV, Karpov P, Van Deursen R, Godin G (2020) State-of-the-art augmented NLP transformer models for direct and single-step retrosynthesis. Nat Commun. https:\/\/doi.org\/10.1038\/s41467-020-19266-y","journal-title":"Nat Commun"},{"key":"2177_CR54","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.4120317","author":"M Barbella","year":"2022","unstructured":"Barbella M, Tortora G (2022) Rouge metric evaluation for text summarization techniques. SSRN Electron J. https:\/\/doi.org\/10.2139\/ssrn.4120317","journal-title":"SSRN Electron J"},{"key":"2177_CR55","doi-asserted-by":"crossref","unstructured":"Li X, Wang W, Hu X, Yang J (2019) Selective kernel networks. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR.2019.00060"},{"key":"2177_CR56","doi-asserted-by":"crossref","unstructured":"Zhang Z, Zhang H, Zhao L et al (2022) Nested hierarchical transformer: towards accurate, data-efficient and interpretable visual understanding. In: Proceedings of the 36th AAAI conference on artificial intelligence, AAAI 2022","DOI":"10.1609\/aaai.v36i3.20252"},{"key":"2177_CR57","doi-asserted-by":"crossref","unstructured":"Yuan K, Guo S, Liu Z et al (2021) Incorporating convolution designs into visual transformers. In: Proceedings of the IEEE international conference on computer vision","DOI":"10.1109\/ICCV48922.2021.00062"},{"key":"2177_CR58","doi-asserted-by":"crossref","unstructured":"Konstantinidis D, Papastratis I, Dimitropoulos KPD (2022) Multi-manifold atten vis transform","DOI":"10.1109\/ACCESS.2023.3329952"},{"key":"2177_CR59","unstructured":"Dagli R (2023) Astroformer: more data might not be all you need for classification. In: ICLR 2023"},{"key":"2177_CR60","doi-asserted-by":"crossref","unstructured":"Liu J, Wen D, Wang D et al (2020) QuantNet: learning to quantize by learning within fully differentiable framework. In: Lecture notes in computer science (including subseries lecture notes in artificial intelligence and lecture notes in bioinformatics)","DOI":"10.1007\/978-3-030-68238-5_4"},{"key":"2177_CR61","unstructured":"Belhasin O, Bar-Shalom GRE-Y (2022) TransBoost: improving the best imagenet performance using deep transductionle. In: 36th conference on neural information processing systems"},{"key":"2177_CR62","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2204.12196","author":"Z Su","year":"2022","unstructured":"Su Z, Zhang H, Chen J, Pang L, Chong-Wah-Ngo Y-GJ (2022) Adaptive split-fusion transformer. Comput Vis Pattern Recognit. https:\/\/doi.org\/10.48550\/arXiv.2204.12196","journal-title":"Comput Vis Pattern Recognit"},{"key":"2177_CR63","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2020.09.037","author":"R Bakalo","year":"2021","unstructured":"Bakalo R, Goldberger J, Ben-Ari R (2021) Weakly and semi supervised detection in medical imaging via deep dual branch net. Neurocomputing. https:\/\/doi.org\/10.1016\/j.neucom.2020.09.037","journal-title":"Neurocomputing"},{"key":"2177_CR64","doi-asserted-by":"publisher","DOI":"10.1145\/3177745","author":"M Cornia","year":"2018","unstructured":"Cornia M, Baraldi L, Serra G, Cucchiara R (2018) Paying more attention to saliency: image captioning with saliency and context attention. ACM Trans Multimed Comput Commun Appl. https:\/\/doi.org\/10.1145\/3177745","journal-title":"ACM Trans Multimed Comput Commun Appl"},{"key":"2177_CR65","doi-asserted-by":"crossref","unstructured":"Zhou L, Palangi H, Zhang L et al (2020) Unified vision-language pre-training for image captioning and VQA. In: AAAI 2020-34th AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v34i07.7005"},{"key":"2177_CR66","unstructured":"Hao Y, Song H, Dong L, Huang S, Chi Z, Wang W, Shuming Ma FW (2022) Language models are general-purpose interfaces"},{"key":"2177_CR67","doi-asserted-by":"crossref","unstructured":"Jin W, Cheng Y, Shen Y et al (2022) A good prompt is worth millions of parameters: low-resource prompt-based learning for vision-language models. In: Proceedings of the annual meeting of the association for computational linguistics","DOI":"10.18653\/v1\/2022.acl-long.197"},{"key":"2177_CR68","unstructured":"Cho J, Lei J, Tan T, Bansal M (2021) Unifying vision-and-language tasks via text generation. In: ICML, pp 1931\u20131942"},{"key":"2177_CR69","unstructured":"Muneeb ul Hassan (2018) VGG16-convolutional network for classification and detection. Neurohive"},{"key":"2177_CR70","unstructured":"Tan M, Le QV (2021) EfficientNetV2: smaller models and faster training. In: Proceedings of machine learning research"},{"key":"2177_CR71","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings-30th IEEE conference on computer vision and pattern recognition, CVPR 2017","DOI":"10.1109\/CVPR.2017.243"},{"key":"2177_CR72","unstructured":"Ekoputris RO (2018) MobileNet: Deteksi Objek pada Platform Mobile | by Rizqi Okta Ekoputris | Nodeflux | Medium. In: 9 May 2018"},{"key":"2177_CR73","doi-asserted-by":"crossref","unstructured":"Liu Z, Mao H, Wu CY et al (2022) A ConvNet for the 2020s. In: Proceedings of the IEEE computer society conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"2177_CR74","unstructured":"Goodfellow IJ, Shlens J, Szegedy C (2015) Explaining and harnessing adversarial examples. In: 3rd international conference on learning representations, ICLR 2015-conference track proceedings"},{"key":"2177_CR75","unstructured":"Borkowski AA, Bui MM, Thomas LB et al (2019) Lung and colon cancer histopathological image dataset (LC25000)"},{"key":"2177_CR76","unstructured":"BreakHis [OL]. https:\/\/www.kaggle.com\/datasets\/ambarish\/breakhis"},{"key":"2177_CR77","doi-asserted-by":"publisher","DOI":"10.1016\/j.cell.2018.02.010","author":"DS Kermany","year":"2018","unstructured":"Kermany DS, Goldbaum M, Cai W et al (2018) Identifying medical diagnoses and treatable diseases by image-based deep learning. Cell. https:\/\/doi.org\/10.1016\/j.cell.2018.02.010","journal-title":"Cell"},{"key":"2177_CR78","unstructured":"Retinal OCT Images (optical coherence tomography)tle. https:\/\/www.kaggle.com\/datasets\/paultimothymooney\/kermany2018. Accessed 31 Oct 2023"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02177-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-024-02177-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02177-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,13]],"date-time":"2024-09-13T16:41:01Z","timestamp":1726245661000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-024-02177-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,17]]},"references-count":78,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2024,10]]}},"alternative-id":["2177"],"URL":"https:\/\/doi.org\/10.1007\/s13042-024-02177-5","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5,17]]},"assertion":[{"value":"21 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 April 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 May 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}