{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T05:28:44Z","timestamp":1730266124697,"version":"3.28.0"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,18]],"date-time":"2023-06-18T00:00:00Z","timestamp":1687046400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,18]],"date-time":"2023-06-18T00:00:00Z","timestamp":1687046400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,18]]},"DOI":"10.1109\/ijcnn54540.2023.10191558","type":"proceedings-article","created":{"date-parts":[[2023,8,2]],"date-time":"2023-08-02T17:30:03Z","timestamp":1690997403000},"page":"1-8","source":"Crossref","is-referenced-by-count":1,"title":["Towards Few-shot Image Captioning with Cycle-based Compositional Semantic Enhancement Framework"],"prefix":"10.1109","author":[{"given":"Peng","family":"Zhang","sequence":"first","affiliation":[{"name":"Durham University,Department of Computer Science,Durham,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Bai","sequence":"additional","affiliation":[{"name":"Durham University,Department of Computer Science,Durham,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Su","sequence":"additional","affiliation":[{"name":"Newcasle University,Department of Computer Science,Newcasle,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Huang","sequence":"additional","affiliation":[{"name":"Institute of Automation Chinese Academy of Sciences,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Long","sequence":"additional","affiliation":[{"name":"Durham University,Department of Computer Science,Durham,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.1109\/CVPR42600.2020.01059"},{"doi-asserted-by":"publisher","key":"ref35","DOI":"10.1109\/ICCV.2019.00473"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1609\/aaai.v35i3.16328"},{"key":"ref34","article-title":"Image captioning: Transforming objects into words","author":"herdade","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref15","first-page":"19822","article-title":"Cogview: Mastering text-to-image generation via transformers","volume":"34","author":"ding","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref37","article-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","author":"ren","year":"2015","journal-title":"Advances in neural information processing systems"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.1109\/CVPR.2018.00754"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.1109\/CVPR.2015.7298932"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1109\/CVPR.2018.00636"},{"key":"ref30","first-page":"7008","article-title":"Self-critical sequence training for image captioning","author":"steven","year":"0","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.21236\/ADA623249"},{"key":"ref33","first-page":"684","article-title":"Exploring visual relationship for image captioning","author":"yao","year":"0","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref10","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1007\/978-3-030-01216-8_31"},{"key":"ref2","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"radford","year":"0","journal-title":"International Conference on Machine Learning"},{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.1109\/TPAMI.2017.2762295"},{"key":"ref17","article-title":"Vinvl: Making visual representations matter in vision-language models","author":"zhang","year":"2021","journal-title":"CVPR 2021"},{"doi-asserted-by":"publisher","key":"ref39","DOI":"10.1007\/s11263-016-0981-7"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1007\/978-3-030-58577-8_8"},{"doi-asserted-by":"publisher","key":"ref38","DOI":"10.1109\/CVPR.2016.90"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1109\/CVPR.2017.130"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/CVPR.2016.8"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.1109\/CVPR52688.2022.01805"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.1007\/978-3-031-19778-9_40"},{"key":"ref26","first-page":"65","article-title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","author":"banerjee","year":"0","journal-title":"Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.3115\/1073083.1073135"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.1109\/TCSVT.2020.2965966"},{"doi-asserted-by":"publisher","key":"ref22","DOI":"10.1109\/ICCV.2019.00751"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1109\/CVPR.2019.00425"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"ref27","first-page":"74","article-title":"Rouge: A package for automatic evaluation of summaries","author":"lin","year":"2004","journal-title":"Text Summarization Branches Out"},{"key":"ref29","article-title":"How much can clip benefit vision-and-language tasks?","author":"shen","year":"2021","journal-title":"ArXiv Preprint"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1007\/978-3-031-19772-7_25"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.1145\/3394171.3414064"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1109\/CVPR.2019.01283"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.1609\/aaai.v33i01.33018489"},{"key":"ref3","first-page":"8821","article-title":"Zero-shot text-to-image generation","author":"ramesh","year":"0","journal-title":"International Conference on Machine Learning"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1145\/3474085.3475519"},{"key":"ref5","article-title":"Deep captioning with multimodal recurrent neural networks (m-rnn)","author":"mao","year":"2014","journal-title":"ArXiv Preprint"}],"event":{"name":"2023 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2023,6,18]]},"location":"Gold Coast, Australia","end":{"date-parts":[[2023,6,23]]}},"container-title":["2023 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10190990\/10190992\/10191558.pdf?arnumber=10191558","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,21]],"date-time":"2023-08-21T17:45:24Z","timestamp":1692639924000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10191558\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,18]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/ijcnn54540.2023.10191558","relation":{},"subject":[],"published":{"date-parts":[[2023,6,18]]}}}