{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T05:33:38Z","timestamp":1730266418980,"version":"3.28.0"},"reference-count":38,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,6,30]],"date-time":"2024-06-30T00:00:00Z","timestamp":1719705600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,6,30]],"date-time":"2024-06-30T00:00:00Z","timestamp":1719705600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,6,30]]},"DOI":"10.1109\/ijcnn60899.2024.10650474","type":"proceedings-article","created":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T17:35:05Z","timestamp":1725903305000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Look and Review, Then Tell: Generate More Coherent Paragraphs from Images by Fusing Visual and Textual Information"],"prefix":"10.1109","author":[{"given":"Zhen","family":"Yang","sequence":"first","affiliation":[{"name":"Beijing Institute of Technology,School of Computer Science and Technology,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongxia","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Automation Academy of Sciences,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ping","family":"Jian","sequence":"additional","affiliation":[{"name":"Beijing Institute of Technology,School of Computer Science and Technology,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.356"},{"article-title":"Vqa4cir: Boosting composed image retrieval with visual question answering","year":"2023","author":"Feng","key":"ref2"},{"key":"ref3","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021","volume":"139","author":"Radford"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"70","DOI":"10.1007\/978-3-031-19836-6_5","article-title":"Storydall-e: Adapting pretrained text-to-image transformers for story continuation","volume-title":"Computer Vision - ECCV 2022 - 17th European Conference","volume":"13697","author":"Maharana"},{"article-title":"Radialog: A large vision-language model for radiology report generation and conversational assistance","year":"2023","author":"Pellegrini","key":"ref5"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.5555\/3045118.3045336"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_45"},{"article-title":"Paracnn: Visual paragraph generation via adversarial twin contextual cnns","year":"2020","author":"Yan","key":"ref8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.279"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01624-6"},{"key":"ref11","first-page":"530","article-title":"Mutual information neural estimation","volume-title":"Proceedings of the 35th International Conference on Machine Learning, ICML 2018","volume":"80","author":"Belghazi"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICME52920.2022.9859701"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240583"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1084"},{"article-title":"Learning deep representations by mutual information estimation and maximization","volume-title":"7th International Conference on Learning Representations, ICLR 2019","author":"Hjelm","key":"ref15"},{"key":"ref16","first-page":"687","article-title":"Prompt-based logical semantics enhancement for implicit discourse relation recognition","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, EMNLP 2023","author":"Wang"},{"key":"ref17","article-title":"Augmenting convolutional networks with attention-based aggregation","volume-title":"CoRR","author":"Touvron","year":"2021"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.364"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.323"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i18.17892"},{"key":"ref21","article-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","volume":"28","author":"Ren","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.4142431"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.499"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00302"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/p15-2017"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1017\/S1351324918000098"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"ref28","first-page":"65","article-title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","author":"Banerjee"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.494"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00636"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICARM49381.2020.9195335"},{"article-title":"Enhancing image captioning with depth information using a transformer-based framework","year":"2023","author":"Ahmed","key":"ref35"},{"article-title":"Adam: A method for stochastic optimization","volume-title":"3rd International Conference on Learning Representations, ICLR 2015","author":"Kingma","key":"ref36"},{"key":"ref37","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2023.findings-ijcnlp.32","article-title":"Exploring the use of large language models for reference-free text quality evaluation: A preliminary empirical study","volume-title":"CoRR","author":"Chen","year":"2023"},{"key":"ref38","article-title":"Judging llm-as-a-judge with mt-bench and chatbot arena","volume-title":"CoRR","author":"Zheng","year":"2023"}],"event":{"name":"2024 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2024,6,30]]},"location":"Yokohama, Japan","end":{"date-parts":[[2024,7,5]]}},"container-title":["2024 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10649807\/10649898\/10650474.pdf?arnumber=10650474","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T05:22:27Z","timestamp":1725945747000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10650474\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,30]]},"references-count":38,"URL":"https:\/\/doi.org\/10.1109\/ijcnn60899.2024.10650474","relation":{},"subject":[],"published":{"date-parts":[[2024,6,30]]}}}