{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T10:58:23Z","timestamp":1743073103856,"version":"3.40.3"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030454388"},{"type":"electronic","value":"9783030454395"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-45439-5_4","type":"book-chapter","created":{"date-parts":[[2020,4,11]],"date-time":"2020-04-11T04:02:50Z","timestamp":1586577770000},"page":"50-64","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Variational Recurrent Sequence-to-Sequence Retrieval for Stepwise Illustration"],"prefix":"10.1007","author":[{"given":"Vishwash","family":"Batra","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aparajita","family":"Haldar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yulan","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hakan","family":"Ferhatosmanoglu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"George","family":"Vogiatzis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tanaya","family":"Guha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,4,8]]},"reference":[{"key":"4_CR1","unstructured":"Alikhani, M., Chowdhury, S.N., de Melo, G., Stone, M.: CITE: a corpus of image-text discourse relations. arXiv preprint arXiv:1904.06286 (2019)"},{"key":"4_CR2","doi-asserted-by":"crossref","unstructured":"Balaneshin-kordan, S., Kotov, A.: Deep neural architecture for multi-modal retrieval based on joint embedding space for text and images. In: Proceedings of the Eleventh ACM International Conference on Web Search and Data Mining, pp. 28\u201336. ACM (2018)","DOI":"10.1145\/3159652.3159735"},{"key":"4_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"446","DOI":"10.1007\/978-3-319-10599-4_29","volume-title":"Computer Vision \u2013 ECCV 2014","author":"L Bossard","year":"2014","unstructured":"Bossard, L., Guillaumin, M., Van Gool, L.: Food-101 \u2013 mining discriminative components with random forests. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8694, pp. 446\u2013461. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10599-4_29"},{"key":"4_CR4","doi-asserted-by":"crossref","unstructured":"Carvalho, M., Cad\u00e8ne, R., Picard, D., Soulier, L., Thome, N., Cord, M.: Cross-modal retrieval in the cooking context: learning semantic text-image embeddings. In: The 41st International ACM SIGIR Conference on Research & Development in Information Retrieval, pp. 35\u201344. ACM (2018)","DOI":"10.1145\/3209978.3210036"},{"key":"4_CR5","doi-asserted-by":"publisher","unstructured":"Chandu, K., Nyberg, E., Black, A.W.: Storyboarding of recipes: grounded contextual generation. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 6040\u20136046. Association for Computational Linguistics, Florence, Italy, July 2019. https:\/\/doi.org\/10.18653\/v1\/P19-1606. https:\/\/www.aclweb.org\/anthology\/P19-1606","DOI":"10.18653\/v1\/P19-1606"},{"key":"4_CR6","unstructured":"Chung, J., Kastner, K., Dinh, L., Goel, K., Courville, A.C., Bengio, Y.: A recurrent latent variable model for sequential data. In: Advances in Neural Information Processing Systems, pp. 2980\u20132988 (2015)"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"4_CR8","unstructured":"Faghri, F., Fleet, D.J., Kiros, J.R., Fidler, S.: VSE++: improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612 (2017)"},{"key":"4_CR9","doi-asserted-by":"crossref","unstructured":"Feng, F., Wang, X., Li, R.: Cross-modal retrieval with correspondence autoencoder. In: Proceedings of the 22nd ACM International Conference on Multimedia, pp. 7\u201316. ACM (2014)","DOI":"10.1145\/2647868.2654902"},{"key":"4_CR10","unstructured":"Feng, Y., Lapata, M.: Topic models for image annotation and text illustration. In: Human Language Technologies: The 2010 Annual Conference of the North American Chapter of the Association for Computational Linguistics (HLT 2010), pp. 831\u2013839. Association for Computational Linguistics, Stroudsburg, PA, USA (2010). http:\/\/dl.acm.org\/citation.cfm?id=1857999.1858124"},{"key":"4_CR11","doi-asserted-by":"crossref","unstructured":"Goldberg, A.B., Zhu, X., Dyer, C.R., Eldawy, M., Heng, L.: Easy as ABC?: facilitating pictorial communication via semantically enhanced layout. In: Proceedings of the Twelfth Conference on Computational Natural Language Learning (CoNLL 2008), pp. 119\u2013126. Association for Computational Linguistics, Stroudsburg, PA, USA (2008). http:\/\/dl.acm.org\/citation.cfm?id=1596324.1596345","DOI":"10.3115\/1596324.1596345"},{"issue":"12","key":"4_CR12","doi-asserted-by":"publisher","first-page":"2639","DOI":"10.1162\/0899766042321814","volume":"16","author":"DR Hardoon","year":"2004","unstructured":"Hardoon, D.R., Szedmak, S., Shawe-Taylor, J.: Canonical correlation analysis: an overview with application to learning methods. Neural Comput. 16(12), 2639\u20132664 (2004)","journal-title":"Neural Comput."},{"issue":"7","key":"4_CR13","doi-asserted-by":"publisher","first-page":"1363","DOI":"10.1109\/TMM.2016.2558463","volume":"18","author":"Y He","year":"2016","unstructured":"He, Y., Xiang, S., Kang, C., Wang, J., Pan, C.: Cross-modal retrieval via deep and bidirectional representation learning. IEEE Trans. Multimed. 18(7), 1363\u20131377 (2016)","journal-title":"IEEE Trans. Multimed."},{"key":"4_CR14","unstructured":"Huang, T.H.K., et al.: Visual storytelling. In: Proceedings of the 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 1233\u20131239 (2016)"},{"issue":"1","key":"4_CR15","doi-asserted-by":"publisher","first-page":"68","DOI":"10.1145\/1126004.1126008","volume":"2","author":"D Joshi","year":"2006","unstructured":"Joshi, D., Wang, J.Z., Li, J.: The story picturing engine\u2013a system for automatic text illustration. ACM Trans. Multimed. Comput. Commun. Appl. 2(1), 68\u201389 (2006). https:\/\/doi.org\/10.1145\/1126004.1126008","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"4","key":"4_CR16","doi-asserted-by":"publisher","first-page":"664","DOI":"10.1109\/TPAMI.2016.2598339","volume":"39","author":"A Karpathy","year":"2017","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. IEEE Trans. Pattern Anal. Mach. Intell. 39(4), 664\u2013676 (2017). https:\/\/doi.org\/10.1109\/TPAMI.2016.2598339","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4_CR17","unstructured":"Kim, G., Moon, S., Sigal, L.: Ranking and retrieval of image sequences from multiple paragraph queries. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1993\u20132001 (2015)"},{"key":"4_CR18","unstructured":"Kiros, R., Salakhutdinov, R., Zemel, R.S.: Unifying visual-semantic embeddings with multimodal neural language models. CoRR abs\/1411.2539 (2014). http:\/\/arxiv.org\/abs\/1411.2539"},{"key":"4_CR19","unstructured":"Lin, T., et al.: Microsoft COCO: common objects in context. CoRR abs\/1405.0312 (2014). http:\/\/arxiv.org\/abs\/1405.0312"},{"key":"4_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Y., Fu, J., Mei, T., Chen, C.W.: Let your photos talk: generating narrative paragraph for photo stream via bidirectional attention recurrent neural networks. In: Thirty-First AAAI Conference on Artificial Intelligence (2017)","DOI":"10.1609\/aaai.v31i1.10760"},{"issue":"Nov","key":"4_CR21","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(Nov), 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"4_CR22","unstructured":"Marin, J., et al.: Recipe1M+: a dataset for learning cross-modal embeddings for cooking recipes and food images. arXiv preprint arXiv:1810.06553 (2018)"},{"key":"4_CR23","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., Ng, A.Y.: Multimodal deep learning. In: Proceedings of the 28th International Conference on Machine Learning (ICML 2011), pp. 689\u2013696 (2011)"},{"key":"4_CR24","unstructured":"Park, C.C., Kim, G.: Expressing an image stream with a sequence of natural sentences. In: Advances in Neural Information Processing Systems, pp. 73\u201381 (2015)"},{"key":"4_CR25","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: 2015 IEEE International Conference on Computer Vision (ICCV), pp. 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"4_CR26","unstructured":"Quadrianto, N., Lampert, C.: Learning multi-view neighborhood preserving projections. In: Proceedings of the 28th International Conference on Machine Learning, Washington, USA, 28 June\u20132 July 2011, pp. 425\u2013432. Association for Computing Machinery (2011)"},{"key":"4_CR27","doi-asserted-by":"crossref","unstructured":"Ravi, H., Wang, L., Muniz, C., Sigal, L., Metaxas, D., Kapadia, M.: Show me a story: towards coherent neural story illustration. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7613\u20137621 (2018)","DOI":"10.1109\/CVPR.2018.00794"},{"key":"4_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1007\/11752790_2","volume-title":"Subspace, Latent Structure and Feature Selection","author":"R Rosipal","year":"2006","unstructured":"Rosipal, R., Kr\u00e4mer, N.: Overview and recent advances in partial least squares. In: Saunders, C., Grobelnik, M., Gunn, S., Shawe-Taylor, J. (eds.) SLSFS 2005. LNCS, vol. 3940, pp. 34\u201351. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11752790_2"},{"key":"4_CR29","doi-asserted-by":"crossref","unstructured":"Salvador, A., et al.: Learning cross-modal embeddings for cooking recipes and food images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3020\u20133028 (2017)","DOI":"10.1109\/CVPR.2017.327"},{"key":"4_CR30","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2556\u20132565. Association for Computational Linguistics (2018). http:\/\/aclweb.org\/anthology\/P18-1238","DOI":"10.18653\/v1\/P18-1238"},{"key":"4_CR31","doi-asserted-by":"crossref","unstructured":"Su, J., Wu, S., Xiong, D., Lu, Y., Han, X., Zhang, B.: Variational recurrent neural machine translation. In: Thirty-Second AAAI Conference on Artificial Intelligence (2018)","DOI":"10.1609\/aaai.v32i1.11985"},{"key":"4_CR32","doi-asserted-by":"crossref","unstructured":"Wang, J., He, Y., Kang, C., Xiang, S., Pan, C.: Image-text cross-modal retrieval via modality-specific feature learning. In: Proceedings of the 5th ACM on International Conference on Multimedia Retrieval, pp. 347\u2013354. ACM (2015)","DOI":"10.1145\/2671188.2749341"},{"issue":"1","key":"4_CR33","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1007\/s00778-015-0391-4","volume":"25","author":"W Wang","year":"2015","unstructured":"Wang, W., Yang, X., Ooi, B.C., Zhang, D., Zhuang, Y.: Effective deep learning-based multi-modal retrieval. VLDB J. 25(1), 79\u2013101 (2015). https:\/\/doi.org\/10.1007\/s00778-015-0391-4","journal-title":"VLDB J."},{"key":"4_CR34","doi-asserted-by":"crossref","unstructured":"Yagcioglu, S., Erdem, A., Erdem, E., Ikizler-Cinbis, N.: RecipeQA: a challenge dataset for multimodal comprehension of cooking recipes. arXiv preprint arXiv:1809.00812 (2018)","DOI":"10.18653\/v1\/D18-1166"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-45439-5_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,13]],"date-time":"2024-03-13T19:12:15Z","timestamp":1710357135000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-45439-5_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030454388","9783030454395"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-45439-5_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"8 April 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lisbon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 April 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 April 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"42","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2020.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"457","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"55","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"46","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"12% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Also included: 8 reproducibility papers, 10 demonstration papers, 12 CLEF organizers lab track papers, 7 doctoral consortium papers, 4 workshops, 3 tutorials. Due to the COVID-19 pandemic, this conference was held virtually.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}