{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:04:02Z","timestamp":1784358242542,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819235124","type":"print"},{"value":"9789819235131","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3513-1_28","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T06:40:42Z","timestamp":1784356842000},"page":"335-344","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["FineCap: Hierarchical Progressive Prefix Decoding with Decoupled Attention for Fine-Grained Image Captioning"],"prefix":"10.1007","author":[{"given":"Chenxi","family":"Ge","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junlin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"28_CR1","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., Erhan, D.: Show and tell: a neural image caption generator. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"28_CR2","doi-asserted-by":"crossref","unstructured":"Fang, H., et al.: Platt and others: from captions to visual concepts and back. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1473\u20131482 (2015)","DOI":"10.1109\/CVPR.2015.7298754"},{"key":"28_CR3","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"28_CR4","doi-asserted-by":"crossref","unstructured":"Li, X., et al.: Oscar: object-semantics aligned pre-training for vision-language tasks. In: European Conference on Computer Vision, pp. 121\u2013137 (2020)","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"28_CR5","doi-asserted-by":"crossref","unstructured":"Zhou, L., Palangi, H., Zhang, L., Hu, H., Corso, J., Gao, J.: Unified vision-language pre-training for image captioning and vqa. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 13041\u201313049 (2020)","DOI":"10.1609\/aaai.v34i07.7005"},{"key":"28_CR6","doi-asserted-by":"crossref","unstructured":"Zhang, P., et al.: Vinvl: revisiting visual representations in vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5579\u20135588 (2021)","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"28_CR7","unstructured":"Radford, A., et al.: Clark and others: learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"28_CR8","unstructured":"Radford, A., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1, 9 (2019)"},{"key":"28_CR9","doi-asserted-by":"crossref","unstructured":"Su, Y., et al.: Agentic-SQL taxonomy: a survey of autonomous and interactive text-to-SQL with LLMs (2026)","DOI":"10.22541\/au.177430005.57777158\/v2"},{"key":"28_CR10","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"28_CR11","doi-asserted-by":"crossref","unstructured":"Li, X.L., Liang, P.: Prefix-tuning: optimizing continuous prompts for generation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (volume 1: Long papers), pp. 4582\u20134597 (2021)","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"28_CR12","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., et al.: Microsoft coco: common objects in context. In: European Conference on Computer Vision, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"28_CR13","doi-asserted-by":"crossref","unstructured":"Agrawal, H., et al: Nocaps: novel object captioning at scale. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8948\u20138957 (2019)","DOI":"10.1109\/ICCV.2019.00904"},{"key":"28_CR14","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (volume 1: Long papers), pp. 2556\u20132565 (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"28_CR15","unstructured":"Wang, Z., et al.: SimVLM: simple visual language model pretraining with weak supervision. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"28_CR16","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.-W., Lee K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, volume 1 (long and short papers), pp. 4171\u20134186 (2019)","DOI":"10.18653\/v1\/N19-1423"},{"key":"28_CR17","unstructured":"Radford, A., et al.: Improving language understanding by generative pre-training. OpenAI Blog (2018)"},{"key":"28_CR18","doi-asserted-by":"crossref","unstructured":"Lu, Y., Hu K., Zhang, L.: S3G: stock state space graph for enhanced stock trend prediction. In: ICASSP 2026-2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4081\u20134085 (2026)","DOI":"10.1109\/ICASSP55912.2026.11463578"},{"key":"28_CR19","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"28_CR20","doi-asserted-by":"crossref","unstructured":"Anderson, P., Fernando, B., Johnson M., Gould, S.: Spice: Semantic propositional image caption evaluation. In: European Conference on Computer Vision, pp. 382\u2013398 (2016)","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"28_CR21","unstructured":"Wang, J., Fan, L., Li B., Zhang, L.: Forecasting with guidance: Representation-level supervision for time series forecasting (2026). arXiv preprint arXiv:2603.24262"},{"key":"28_CR22","doi-asserted-by":"crossref","unstructured":"Z. Guo, K. Zhao and L. Zhang: InstanceRSR: real-world super-resolution via instanceaware representation alignment. In: ICASSP 2026-2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 10577\u201310581 (2026)","DOI":"10.1109\/ICASSP55912.2026.11462690"},{"key":"28_CR23","doi-asserted-by":"crossref","unstructured":"Chen, Y.-C., et al.: UNITER: universal image-text representation learning. In: European Conference on Computer Vision (ECCV), pp. 104\u2013120 (2020)","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"28_CR24","doi-asserted-by":"crossref","unstructured":"Wang, J., Fan, L., Li, B., Zhang, L.: A dynamic factor gating architecture with market regime awareness for stock return forecasting. Preprints (2026)","DOI":"10.20944\/preprints202603.2262.v1"},{"key":"28_CR25","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"28_CR26","doi-asserted-by":"crossref","unstructured":"Denkowski, M., Lavie, A.: Meteor universal: language specific translation evaluation for any target language. In: Proceedings of the Ninth Workshop on Statistical Machine Translation, pp. 376\u2013380 (2014)","DOI":"10.3115\/v1\/W14-3348"},{"key":"28_CR27","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization (2014). arXiv preprint arXiv:1412.6980"},{"key":"28_CR28","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900 (2022)"},{"key":"28_CR29","unstructured":"Mokady, R., Hertz A., Bermano, A.H.: Clipcap: clip prefix for image captioning. In: (2021). arXiv preprint arXiv:2111.09734"},{"key":"28_CR30","doi-asserted-by":"crossref","unstructured":"Huang, L., Wang, W., Chen, J., Wei, X.-Y.: Attention on attention for image captioning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4634\u20134643 (2019)","DOI":"10.1109\/ICCV.2019.00473"},{"key":"28_CR31","doi-asserted-by":"crossref","unstructured":"Nguyen, V.-Q., Suganuma, M., Okatani, T.: Grit: faster and better image captioning transformer using dual visual features. In: European Conference on Computer Vision, pp. 167\u2013184 (2022)","DOI":"10.1007\/978-3-031-20059-5_10"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3513-1_28","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T06:40:56Z","timestamp":1784356856000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3513-1_28"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819235124","9789819235131"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3513-1_28","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}