{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T18:58:57Z","timestamp":1785265137010,"version":"3.55.0"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s11760-026-05543-8","type":"journal-article","created":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:49:24Z","timestamp":1783151364000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["DAME-Cap: a dual-teacher and parallel attention\u2013mambablock framework for image captioning"],"prefix":"10.1007","volume":"20","author":[{"given":"Kangzhen","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Juan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongbin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhijun","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,4]]},"reference":[{"key":"5543_CR1","doi-asserted-by":"crossref","unstructured":"Zhou, L., Palangi, H., Zhang, L., Hu, H., Corso, J., Gao, J.: Unified vision-language pre-training for image captioning and vqa. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 13041\u201313049. (2020)","DOI":"10.1609\/aaai.v34i07.7005"},{"key":"5543_CR2","unstructured":"Wang, J., Yang, Z., Hu, X., Li, L., Lin, K., Gan, Z., Liu, Z., Liu, C., Wang, L.: Git: A generative image-to-text transformer for vision and language. arXiv preprint arXiv:2205.14100 (2022)"},{"key":"5543_CR3","unstructured":"Gu, A., Goel, K., R\u00e9, C.: Efficiently modeling long sequences with structured state spaces, (2021). arXiv:2111.00396 arXiv preprint"},{"key":"5543_CR4","doi-asserted-by":"publisher","first-page":"16344","DOI":"10.52202\/068431-1189","volume":"35","author":"T Dao","year":"2022","unstructured":"Dao, T., Fu, D., Ermon, S., Rudra, A., R\u00e9, C.: Flashattention: Fast and memory-efficient exact attention with io-awareness. Adv. Neural. Inf. Process. Syst. 35, 16344\u201316359 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5543_CR5","unstructured":"Gu, A., Dao, T.: Mamba: Linear-time sequence modeling with selective state spaces, (2023). arXiv:2312.00752 arXiv preprint"},{"key":"5543_CR6","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network, (2015). arXiv:1503.02531 arXiv preprint"},{"key":"5543_CR7","doi-asserted-by":"crossref","unstructured":"Ahn, S., Hu, S.X., Damianou, A., Lawrence, N.D., Dai, Z.: Variational information distillation for knowledge transfer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9163\u20139171. (2019)","DOI":"10.1109\/CVPR.2019.00938"},{"key":"5543_CR8","doi-asserted-by":"crossref","unstructured":"Tung, F., Mori, G.: Similarity-preserving knowledge distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1365\u20131374. (2019)","DOI":"10.1109\/ICCV.2019.00145"},{"key":"5543_CR9","unstructured":"Tian, Y., Krishnan, D., Isola, P.: Contrastive representation distillation, (2019). arXiv:1910.10699 arXiv preprint"},{"key":"5543_CR10","doi-asserted-by":"crossref","unstructured":"Fang, Z., Wang, J., Hu, X., Wang, L., Yang, Y., Liu, Z.: Compressing visual-linguistic model via knowledge distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1428\u20131438 (2021)","DOI":"10.1109\/ICCV48922.2021.00146"},{"key":"5543_CR11","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Advances in neural information processing systems 30 (2017)"},{"key":"5543_CR12","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S.: An image is worth 16x16 words: Transformers for image recognition at scale, (2020). arXiv:2010.11929 arXiv preprint"},{"key":"5543_CR13","first-page":"572","volume":"34","author":"A Gu","year":"2021","unstructured":"Gu, A., Johnson, I., Goel, K., Saab, K., Dao, T., Rudra, A., R\u00e9, C.: Combining recurrent, convolutional, and continuous-time models with linear state space layers. Adv. Neural. Inf. Process. Syst. 34, 572\u2013585 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5543_CR14","unstructured":"Zhu, L., Liao, B., Zhang, Q., Wang, X., Liu, W., Wang, X.: Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint arXiv:2401.09417 (2024)"},{"key":"5543_CR15","doi-asserted-by":"crossref","unstructured":"Hatamizadeh, A., Kautz, J.: Mambavision: A hybrid mamba-transformer vision backbone. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 25261\u201325270. (2025)","DOI":"10.1109\/CVPR52734.2025.02352"},{"key":"5543_CR16","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., Zhang, L.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086. (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"5543_CR17","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: Learning cross-modality encoder representations from transformers, (2019). arXiv:1908.07490 arXiv preprint","DOI":"10.18653\/v1\/D19-1514"},{"key":"5543_CR18","doi-asserted-by":"crossref","unstructured":"Li, G., Duan, N., Fang, Y., Gong, M., Jiang, D.: Unicoder-vl: A universal encoder for vision and language by cross-modal pre-training. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11336\u201311344. (2020)","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"5543_CR19","doi-asserted-by":"crossref","unstructured":"Nguyen, V.-Q., Suganuma, M., Okatani, T.: Grit: Faster and better image captioning transformer using dual visual features. In: European Conference on Computer Vision, pp. 167\u2013184. Springer (2022)","DOI":"10.1007\/978-3-031-20059-5_10"},{"key":"5543_CR20","unstructured":"Huang, F.: Ofcap: Object-aware fusion for image captioning, p. 2412. (2024). arXiv e-prints"},{"issue":"6","key":"5543_CR21","doi-asserted-by":"publisher","first-page":"11064","DOI":"10.1109\/TNNLS.2025.3531987","volume":"36","author":"N Wang","year":"2025","unstructured":"Wang, N., Cui, Z., Li, A., Lu, Y., Wang, R., Nie, F.: Structured doubly stochastic graph-based clustering. IEEE Transactions on Neural Networks and Learning Systems 36(6), 11064\u201311077 (2025)","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"5543_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2025.110144","volume":"238","author":"N Wang","year":"2026","unstructured":"Wang, N., Cui, Z., Li, A., Wang, R., Nie, F.: Multi-view clustering based on doubly stochastic graph. Signal Process. 238, 110144 (2026)","journal-title":"Signal Process."},{"key":"5543_CR23","doi-asserted-by":"crossref","unstructured":"Wang, N., Cui, Z., Su, Y., Li, A., Xue, Y., Ren, W.: Dynamic and consistent doubly stochastic similarity learning for multi-view and multi-order clustering. Pattern Recogn. , 113698 (2026)","DOI":"10.1016\/j.patcog.2026.113698"},{"issue":"2","key":"5543_CR24","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1007\/s11063-024-11527-x","volume":"56","author":"Q Sun","year":"2024","unstructured":"Sun, Q., Zhang, J., Fang, Z., Gao, Y.: Self-enhanced attention for image captioning. Neural Process. Lett. 56(2), 131 (2024)","journal-title":"Neural Process. Lett."},{"key":"5543_CR25","unstructured":"Chen, X., Fang, H., Lin, T.-Y., Vedantam, R., Gupta, S., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco captions: Data collection and evaluation server, (2015). arXiv:1504.00325 arXiv preprint"},{"key":"5543_CR26","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137. (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"5543_CR27","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318. (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"5543_CR28","unstructured":"Banerjee, S., Lavie, A.: Meteor: An automatic metric for mt evaluation with improved correlation with human judgments. In: Proceedings of the Acl Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation And\/or Summarization, pp. 65\u201372. (2005)"},{"key":"5543_CR29","unstructured":"Lin, C.-Y.: Rouge: A package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381. (2004)"},{"key":"5543_CR30","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lawrence Zitnick, C., Parikh, D.: Cider: Consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4566\u20134575. (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"5543_CR31","doi-asserted-by":"crossref","unstructured":"Anderson, P., Fernando, B., Johnson, M., Gould, S.: Spice: Semantic propositional image caption evaluation. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part V 14, pp. 382\u2013398 (2016). Springer","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"5543_CR32","unstructured":"Radford, A., Kim, J.W., et al: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (ICML) (2021). urlhttps:\/\/arxiv.org\/abs\/2103.00020"},{"key":"5543_CR33","doi-asserted-by":"crossref","unstructured":"Rennie, S.J., Marcheret, E., Mroueh, Y., Ross, J., Goel, V.: Self-critical sequence training for image captioning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7008\u20137024. (2017)","DOI":"10.1109\/CVPR.2017.131"},{"key":"5543_CR34","unstructured":"Herdade, S., Kappeler, A., Boakye, K., Soares, J.: Image captioning: Transforming objects into words. Adv. Neural. Inf. Process. Syst. 32, (2019)"},{"key":"5543_CR35","doi-asserted-by":"crossref","unstructured":"Yang, X., Tang, K., Zhang, H., Cai, J.: Auto-encoding scene graphs for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10685\u201310694. (2019)","DOI":"10.1109\/CVPR.2019.01094"},{"key":"5543_CR36","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W., Chen, J., Wei, X.-Y.: Attention on attention for image captioning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4634\u20134643. (2019)","DOI":"10.1109\/ICCV.2019.00473"},{"key":"5543_CR37","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10578\u201310587. (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"5543_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109420","volume":"138","author":"Y Ma","year":"2023","unstructured":"Ma, Y., Ji, J., Sun, X., Zhou, Y., Ji, R.: Towards local visual modeling for image captioning. Pattern Recogn. 138, 109420 (2023)","journal-title":"Pattern Recogn."},{"issue":"8","key":"5543_CR39","doi-asserted-by":"publisher","first-page":"4257","DOI":"10.1109\/TCSVT.2023.3243725","volume":"33","author":"J Zhang","year":"2023","unstructured":"Zhang, J., Xie, Y., Ding, W., Wang, Z.: Cross on cross attention: Deep fusion transformer for image captioning. IEEE Trans. Circuits Syst. Video Technol. 33(8), 4257\u20134268 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"5543_CR40","doi-asserted-by":"crossref","unstructured":"Li, Y., Pan, Y., Yao, T., Mei, T.: Comprehending and ordering semantics for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17990\u201317999. (2022)","DOI":"10.1109\/CVPR52688.2022.01746"},{"key":"5543_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, K., Xie, Y., Wang, Z.: Deep reciprocal learning for image captioning. IEEE Transactions on Circuits and Systems for Video Technology (2025)","DOI":"10.1109\/TCSVT.2025.3539344"},{"key":"5543_CR42","doi-asserted-by":"crossref","unstructured":"Pan, Y., Yao, T., Li, Y., Mei, T.: X-linear attention networks for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10971\u201310980. (2020)","DOI":"10.1109\/CVPR42600.2020.01098"},{"key":"5543_CR43","doi-asserted-by":"crossref","unstructured":"Luo, Y., Ji, J., Sun, X., Cao, L., Wu, Y., Huang, F., Lin, C.-W., Ji, R.: Dual-level collaborative transformer for image captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 2286\u20132293 (2021)","DOI":"10.1609\/aaai.v35i3.16328"},{"key":"5543_CR44","doi-asserted-by":"crossref","unstructured":"Barraco, M., Stefanini, M., Cornia, M., Cascianelli, S., Baraldi, L., Cucchiara, R.: Camel: mean teacher learning for image captioning. In: 2022 26th International Conference on Pattern Recognition (ICPR), pp. 4087\u20134094. IEEE (2022)","DOI":"10.1109\/ICPR56361.2022.9955644"},{"issue":"11","key":"5543_CR45","doi-asserted-by":"publisher","first-page":"11900","DOI":"10.1109\/TCSVT.2024.3425513","volume":"34","author":"L Wang","year":"2024","unstructured":"Wang, L., Chen, H., Liu, Y., Lyu, Y.: Regular constrained multimodal fusion for image captioning. IEEE Trans. Circuits Syst. Video Technol. 34(11), 11900\u201311913 (2024)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05543-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-026-05543-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05543-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T18:17:15Z","timestamp":1785262635000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-026-05543-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":45,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["5543"],"URL":"https:\/\/doi.org\/10.1007\/s11760-026-05543-8","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"24 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 June 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 July 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","label":"Competing interests","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"481"}}