{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,16]],"date-time":"2025-09-16T20:48:45Z","timestamp":1758055725781,"version":"3.44.0"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,5,29]],"date-time":"2025-05-29T00:00:00Z","timestamp":1748476800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,29]],"date-time":"2025-05-29T00:00:00Z","timestamp":1748476800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00530-025-01854-x","type":"journal-article","created":{"date-parts":[[2025,5,29]],"date-time":"2025-05-29T06:11:19Z","timestamp":1748499079000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Meta-prompt tuning for low-resource visual question answering"],"prefix":"10.1007","volume":"31","author":[{"given":"Mingwen","family":"Shao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuanyuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingzhuang","family":"Meng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xun","family":"Shao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,29]]},"reference":[{"key":"1854_CR1","doi-asserted-by":"crossref","unstructured":"Antol, S., Agrawal, A., Lu, J., Mitchell, M., Batra, D., Zitnick, C.L., Parikh, D.: Vqa: Visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"1854_CR2","unstructured":"Cho, J., Lei, J., Tan, H., Bansal, M.: Unifying vision-and-language tasks via text generation. In: International Conference on Machine Learning, pp. 1931\u20131942 (2021). PMLR"},{"key":"1854_CR3","doi-asserted-by":"crossref","unstructured":"Yang, Z., Gan, Z., Wang, J., Hu, X., Lu, Y., Liu, Z., Wang, L.: An empirical study of gpt-3 for few-shot knowledge-based vqa. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 3081\u20133089 (2022)","DOI":"10.1609\/aaai.v36i3.20215"},{"key":"1854_CR4","unstructured":"Kim, W., Son, B., Kim, I.: Vilt: Vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning, pp. 5583\u20135594 (2021). PMLR"},{"key":"1854_CR5","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li, J., Selvaraju, R., Gotmare, A., Joty, S., Xiong, C., Hoi, S.C.H.: Align before fuse: Vision and language representation learning with momentum distillation. Adv. Neural. Inf. Process. Syst. 34, 9694\u20139705 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR6","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742 (2023). PMLR"},{"key":"1854_CR7","unstructured":"Wang, P., Yang, A., Men, R., Lin, J., Bai, S., Li, Z., Ma, J., Zhou, C., Zhou, J., Yang, H.: Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: International Conference on Machine Learning, pp. 23318\u201323340 (2022). PMLR"},{"key":"1854_CR8","first-page":"1022","volume":"34","author":"R Karimi Mahabadi","year":"2021","unstructured":"Karimi Mahabadi, R., Henderson, J., Ruder, S.: Compacter: efficient low-rank hypercomplex adapter layers. Adv. Neural. Inf. Process. Syst. 34, 1022\u20131035 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR9","doi-asserted-by":"crossref","unstructured":"Sung, Y.-L., Cho, J., Bansal, M.: Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5227\u20135237 (2022)","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"1854_CR10","doi-asserted-by":"crossref","unstructured":"Jiang, J., Zheng, N.: Mixphm: redundancy-aware parameter-efficient tuning for low-resource visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 24203\u201324213 (2023)","DOI":"10.1109\/CVPR52729.2023.02318"},{"key":"1854_CR11","doi-asserted-by":"crossref","unstructured":"Jin, W., Cheng, Y., Shen, Y., Chen, W., Ren, X.: A good prompt is worth millions of parameters: Low-resource prompt-based learning for vision-language models. arXiv preprint arXiv:2110.08484 (2021)","DOI":"10.18653\/v1\/2022.acl-long.197"},{"key":"1854_CR12","doi-asserted-by":"crossref","unstructured":"Pfeiffer, J., Kamath, A., R\u00fcckl\u00e9, A., Cho, K., Gurevych, I.: Adapterfusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247 (2020)","DOI":"10.18653\/v1\/2021.eacl-main.39"},{"key":"1854_CR13","first-page":"34892","volume":"36","author":"H Liu","year":"2024","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 34892\u201334916 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR14","unstructured":"Finn, C., Abbeel, P., Levine, S.: Model-agnostic meta-learning for fast adaptation of deep networks. In: International Conference on Machine Learning, pp. 1126\u20131135 (2017). PMLR"},{"key":"1854_CR15","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26296\u201326306 (2024)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"1854_CR16","unstructured":"Zhang, A., Tay, Y., Zhang, S., Chan, A., Luu, A.T., Hui, S.C., Fu, J.: Beyond fully-connected layers with quaternions: Parameterization of hypercomplex multiplications with 1\/n parameters. ArXiv arXiv: abs\/2102.08597 (2021)"},{"key":"1854_CR17","doi-asserted-by":"crossref","unstructured":"Huang, Z., Zeng, Z., Huang, Y., Liu, B., Fu, D., Fu, J.: Seeing out of the box: End-to-end pre-training for vision-language representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12976\u201312985 (2021)","DOI":"10.1109\/CVPR46437.2021.01278"},{"key":"1854_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2024.109474","volume":"119","author":"NS Nguyen","year":"2024","unstructured":"Nguyen, N.S., Le, T., et al.: Advancing vietnamese visual question answering with transformer and convolutional integration. Comput. Electr. Eng. 119, 109474 (2024)","journal-title":"Comput. Electr. Eng."},{"key":"1854_CR19","unstructured":"Liu, R., Zhuang, L., Yu, Z., Jiang, Z., Bai, T.: Question-relationship guided graph attention network for visual question answer. Multim. Syst. 1\u201312 (2022)"},{"issue":"5","key":"1854_CR20","doi-asserted-by":"publisher","first-page":"2527","DOI":"10.1007\/s00530-023-01125-7","volume":"29","author":"C Liu","year":"2023","unstructured":"Liu, C., Tan, Y.-Y., Xia, T.-T., Zhang, J., Zhu, M.: Co-attention graph convolutional network for visual question answering. Multim. Syst. 29(5), 2527\u20132543 (2023)","journal-title":"Multim. Syst."},{"key":"1854_CR21","unstructured":"Zaken, E.B., Goldberg, Y., Ravfogel, S.: Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), pp. 1\u20139 (2022)"},{"key":"1854_CR22","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Guo, W., Meng, X., Wang, Y., Wang, Y., Jiang, X., Liu, Q., Yang, Z.: Hyperpelt: Unified parameter-efficient language model tuning for both language and vision-and-language tasks. In: Findings of the Association for Computational Linguistics: ACL 2023, pp. 11442\u201311453 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.725"},{"key":"1854_CR23","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR24","unstructured":"Yang, Z., Dai, Z., Yang, Y., Carbonell, J., Salakhutdinov, R.R., Le, Q.V.: Xlnet: Generalized autoregressive pretraining for language understanding. Adv. Neural Inform. Process. Syst. 32 (2019)"},{"key":"1854_CR25","doi-asserted-by":"publisher","unstructured":"Petroni, F., Rockt\u00e4schel, T., Riedel, S., Lewis, P., Bakhtin, A., Wu, Y., Miller, A.: Language models as knowledge bases? In: Inui, K., Jiang, J., Ng, V., Wan, X. (eds.) Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 2463\u20132473. Association for Computational Linguistics, Hong Kong, China (2019). https:\/\/doi.org\/10.18653\/v1\/D19-1250.https:\/\/aclanthology.org\/D19-1250\/","DOI":"10.18653\/v1\/D19-1250."},{"key":"1854_CR26","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., Constant, N.: The power of scale for parameter-efficient prompt tuning. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 3045\u20133059 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"1854_CR27","doi-asserted-by":"crossref","unstructured":"Li, X.L., Liang, P.: Prefix-tuning: Optimizing continuous prompts for generation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 4582\u20134597 (2021)","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"1854_CR28","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16816\u201316825 (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"5","key":"1854_CR29","doi-asserted-by":"publisher","first-page":"053037","DOI":"10.1117\/1.JEI.32.5.053037","volume":"32","author":"L Zhang","year":"2023","unstructured":"Zhang, L., Shao, M., Chen, S., Liu, F.: Contrastive knowledge-augmented self-distillation approach for few-shot learning. J. Electron. Imag. 32(5), 053037 (2023)","journal-title":"J. Electron. Imag."},{"issue":"5","key":"1854_CR30","doi-asserted-by":"publisher","first-page":"053015","DOI":"10.1117\/1.JEI.33.5.053015","volume":"33","author":"X Shao","year":"2024","unstructured":"Shao, X., Shao, M., Chen, S., Liu, Y.: Usdap: universal source-free domain adaptation based on prompt learning. J. Electron. Imag. 33(5), 053015\u2013053015 (2024)","journal-title":"J. Electron. Imag."},{"key":"1854_CR31","first-page":"200","volume":"34","author":"M Tsimpoukelli","year":"2021","unstructured":"Tsimpoukelli, M., Menick, J.L., Cabi, S., Eslami, S., Vinyals, O., Hill, F.: Multimodal few-shot learning with frozen language models. Adv. Neural. Inf. Process. Syst. 34, 200\u2013212 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR32","first-page":"23716","volume":"35","author":"J-B Alayrac","year":"2022","unstructured":"Alayrac, J.-B., Donahue, J., Luc, P., Miech, A., Barr, I., Hasson, Y., Lenc, K., Mensch, A., Millican, K., Reynolds, M., et al.: Flamingo: a visual language model for few-shot learning. Adv. Neural. Inf. Process. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR33","doi-asserted-by":"crossref","unstructured":"Zhu, L., Wang, X., Ke, Z., Zhang, W., Lau, R.W.: Biformer: Vision transformer with bi-level routing attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10323\u201310333 (2023)","DOI":"10.1109\/CVPR52729.2023.00995"},{"key":"1854_CR34","unstructured":"Li, C., Ge, Y., Li, D., Shan, Y.: Vision-language instruction tuning: a review and analysis. Trans. Mach. Learn. Res. (2023)"},{"key":"1854_CR35","first-page":"55006","volume":"36","author":"C Zhou","year":"2024","unstructured":"Zhou, C., Liu, P., Xu, P., Iyer, S., Sun, J., Mao, Y., Ma, X., Efrat, A., Yu, P., Yu, L., et al.: Lima: less is more for alignment. Adv. Neural. Inf. Process. Syst. 36, 55006\u201355021 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR36","doi-asserted-by":"crossref","unstructured":"Gao, J., Sarkar, B., Xia, F., Xiao, T., Wu, J., Ichter, B., Majumdar, A., Sadigh, D.: Physically grounded vision-language models for robotic manipulation. In: 2024 IEEE International Conference on Robotics and Automation (ICRA), pp. 12462\u201312469 (2024). IEEE","DOI":"10.1109\/ICRA57147.2024.10610090"},{"key":"1854_CR37","unstructured":"Awadalla, A., Gao, I., Gardner, J., Hessel, J., Hanafy, Y., Zhu, W., Marathe, K., Bitton, Y., Gadre, S.Y., Sagawa, S., Jitsev, J., Kornblith, S., Koh, P.W., Ilharco, G., Wortsman, M., Schmidt, L.: Openflamingo: An open-source framework for training large autoregressive vision-language models. ArXiv arxiv: abs\/2308.01390 (2023)"},{"key":"1854_CR38","first-page":"20755","volume":"33","author":"S Baik","year":"2020","unstructured":"Baik, S., Choi, M., Choi, J., Kim, H., Lee, K.M.: Meta-learning with adaptive hyperparameters. Adv. Neural. Inf. Process. Syst. 33, 20755\u201320765 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR39","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the v in vqa matter: Elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"1854_CR40","doi-asserted-by":"crossref","unstructured":"Marino, K., Rastegari, M., Farhadi, A., Mottaghi, R.: Ok-vqa: A visual question answering benchmark requiring external knowledge. In: Proceedings of the IEEE\/cvf Conference on Computer Vision and Pattern Recognition, pp. 3195\u20133204 (2019)","DOI":"10.1109\/CVPR.2019.00331"},{"key":"1854_CR41","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: Gqa: A new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"1854_CR42","unstructured":"Hu, J.E., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Chen, W.: Lora: Low-rank adaptation of large language models. ArXiv arXiv: abs\/2106.09685 (2021)"},{"key":"1854_CR43","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., Morrone, B., De\u00a0Laroussilhe, Q., Gesmundo, A., Attariyan, M., Gelly, S.: Parameter-efficient transfer learning for nlp. In: International Conference on Machine Learning, pp. 2790\u20132799 (2019). PMLR"},{"key":"1854_CR44","doi-asserted-by":"publisher","unstructured":"Wang, Y., Agarwal, S., Mukherjee, S., Liu, X., Gao, J., Awadallah, A.H., Gao, J.: AdaMix: Mixture-of-Adaptations for Parameter-efficient Model Tuning. 2205\u201312410 (2022) https:\/\/doi.org\/10.48550\/arXiv.2205.12410arXiv:2205.12410 [cs.CL]","DOI":"10.48550\/arXiv.2205.12410"},{"key":"1854_CR45","doi-asserted-by":"crossref","unstructured":"Wang, W., Bao, H., Dong, L., Bjorck, J., Peng, Z., Liu, Q., Aggarwal, K., Mohammed, O.K., Singhal, S., Som, S., et al.: Image as a foreign language: Beit pretraining for vision and vision-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19175\u201319186 (2023)","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"1854_CR46","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., Wu, J., Jiang, X., Almeida, D., Wainwright, C., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A., et al.: Training language models to follow instructions with human feedback. Adv. Neural. Inf. Process. Syst. 35, 27730\u201327744 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1854_CR47","doi-asserted-by":"publisher","unstructured":"Najdenkoska, I., Zhen, X., Worring, M.: Meta Learning to Bridge Vision and Language Models for Multimodal Few-Shot Learning. 2302\u201314794 (2023) https:\/\/doi.org\/10.48550\/arXiv.2302.14794arXiv:2302.14794 [cs.CV]","DOI":"10.48550\/arXiv.2302.14794"},{"key":"1854_CR48","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Groth, O., Bernstein, M., Fei-Fei, L.: Visual7w: Grounded question answering in images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4995\u20135004 (2016)","DOI":"10.1109\/CVPR.2016.540"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01854-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01854-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01854-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T09:04:11Z","timestamp":1757927051000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01854-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,29]]},"references-count":48,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["1854"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01854-x","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2025,5,29]]},"assertion":[{"value":"15 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 May 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We declare that there is no Conflict of interest between the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"272"}}