{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T02:28:28Z","timestamp":1777861708210,"version":"3.51.4"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031971402","type":"print"},{"value":"9783031971419","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-97141-9_30","type":"book-chapter","created":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T08:57:49Z","timestamp":1751273869000},"page":"442-456","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["TopicVD: A Topic-Based Dataset of\u00a0Video-Guided Multimodal Machine Translation for\u00a0Documentaries"],"prefix":"10.1007","author":[{"given":"Jinze","family":"Lv","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zi","family":"Long","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xianghua","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,1]]},"reference":[{"issue":"2","key":"30_CR1","first-page":"79","volume":"16","author":"PF Brown","year":"1990","unstructured":"Brown, P.F., et al.: A statistical approach to machine translation. Comput. Linguist. 16(2), 79\u201385 (1990)","journal-title":"Comput. Linguist."},{"issue":"2","key":"30_CR2","first-page":"263","volume":"19","author":"PF Brown","year":"1993","unstructured":"Brown, P.F., Della Pietra, S.A., Della Pietra, V.J., Mercer, R.L.: The mathematics of statistical machine translation: parameter estimation. Comput. Linguist. 19(2), 263\u2013311 (1993)","journal-title":"Comput. Linguist."},{"key":"30_CR3","doi-asserted-by":"crossref","unstructured":"Caglayan, O., Madhyastha, P., Specia, L., Barrault, L.: Probing the need for visual context in multimodal machine translation. arXiv preprint arXiv:1903.08678 (2019)","DOI":"10.18653\/v1\/N19-1422"},{"key":"30_CR4","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"30_CR5","doi-asserted-by":"crossref","unstructured":"Elliott, D., Frank, S., Barrault, L., Bougares, F., Specia, L.: Findings of the second shared task on multimodal machine translation and multilingual image description. arXiv preprint arXiv:1710.07177 (2017)","DOI":"10.18653\/v1\/W17-4718"},{"key":"30_CR6","doi-asserted-by":"crossref","unstructured":"Elliott, D., Frank, S., Sima\u2019an, K., Specia, L.: Multi30k: multilingual English-German image descriptions. arXiv preprint arXiv:1605.00459 (2016)","DOI":"10.18653\/v1\/W16-3210"},{"key":"30_CR7","unstructured":"Elliott, D., K\u00e1d\u00e1r, A.: Imagination improves multimodal translation. arXiv preprint arXiv:1705.04350 (2017)"},{"key":"30_CR8","doi-asserted-by":"crossref","unstructured":"Hitschler, J., Schamoni, S., Riezler, S.: Multimodal pivots for image caption translation. arXiv preprint arXiv:1601.03916 (2016)","DOI":"10.18653\/v1\/P16-1227"},{"key":"30_CR9","unstructured":"Hu, J., et al.: Large multilingual models pivot zero-shot multimodal learning across languages. arXiv preprint arXiv:2308.12038 (2023)"},{"key":"30_CR10","doi-asserted-by":"crossref","unstructured":"Kang, L., et al.: Bigvideo: a large-scale video subtitle translation dataset for multimodal machine translation. arXiv preprint arXiv:2305.18326 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.535"},{"key":"30_CR11","doi-asserted-by":"crossref","unstructured":"Lan, Z., et al.: Exploring better text image translation with multimodal codebook. arXiv preprint arXiv:2305.17415 (2023)","DOI":"10.18653\/v1\/2023.acl-long.192"},{"key":"30_CR12","unstructured":"Li, B., et al.: On vision features in multimodal machine translation. arXiv preprint arXiv:2203.09173 (2022)"},{"key":"30_CR13","doi-asserted-by":"crossref","unstructured":"Li, J., Ataman, D., Sennrich, R.: Vision matters when it should: sanity checking multimodal machine translation models. arXiv preprint arXiv:2109.03415 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.673"},{"key":"30_CR14","doi-asserted-by":"crossref","unstructured":"Li, Y., Shimizu, S., Chu, C., Kurohashi, S., Li, W.: Video-helpful multimodal machine translation. arXiv preprint arXiv:2310.20201 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.260"},{"key":"30_CR15","unstructured":"Li, Y., Shimizu, S., Gu, W., Chu, C., Kurohashi, S.: Visa: an ambiguous subtitles dataset for visual scene-aware machine translation. arXiv preprint arXiv:2201.08054 (2022)"},{"key":"30_CR16","doi-asserted-by":"crossref","unstructured":"Lin, H., et al.: Dynamic context-guided capsule network for multimodal machine translation. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1320\u20131329 (2020)","DOI":"10.1145\/3394171.3413715"},{"key":"30_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"30_CR18","unstructured":"Liu, L., et al.: On the variance of the adaptive learning rate and beyond. arXiv preprint arXiv:1908.03265 (2019)"},{"key":"30_CR19","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"30_CR20","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"30_CR21","unstructured":"Sanabria, R., et al.: How2: a large-scale dataset for multimodal language understanding. arXiv preprint arXiv:1811.00347 (2018)"},{"key":"30_CR22","first-page":"16857","volume":"33","author":"K Song","year":"2020","unstructured":"Song, K., Tan, X., Qin, T., Lu, J., Liu, T.Y.: MPNet: masked and permuted pre-training for language understanding. Adv. Neural. Inf. Process. Syst. 33, 16857\u201316867 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"30_CR23","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1016\/j.ins.2020.11.024","volume":"554","author":"J Su","year":"2021","unstructured":"Su, J., et al.: Multi-modal neural machine translation with deep semantic interactions. Inf. Sci. 554, 47\u201360 (2021)","journal-title":"Inf. Sci."},{"key":"30_CR24","unstructured":"Tang, Z., Zhang, X., Long, Z., Fu, X.: Multimodal neural machine translation with search engine based image retrieval. arXiv preprint arXiv:2208.00767 (2022)"},{"key":"30_CR25","doi-asserted-by":"crossref","unstructured":"Teramen, A., Ohtsuka, T., Kondo, R., Kajiwara, T., Ninomiya, T.: English-to-Japanese multimodal machine translation based on image-text matching of lecture videos. In: Proceedings of the 3rd Workshop on Advances in Language and Vision Research (ALVR), pp. 86\u201391 (2024)","DOI":"10.18653\/v1\/2024.alvr-1.7"},{"key":"30_CR26","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"issue":"2","key":"30_CR27","doi-asserted-by":"publisher","first-page":"515","DOI":"10.1111\/1556-4029.15432","volume":"69","author":"GS Wales","year":"2024","unstructured":"Wales, G.S.: Validation of image stream hashing: a forensic method for content verification. J. Forensic Sci. 69(2), 515\u2013528 (2024)","journal-title":"J. Forensic Sci."},{"key":"30_CR28","doi-asserted-by":"crossref","unstructured":"Wang, X., Wu, J., Chen, J., Li, L., Wang, Y.F., Wang, W.Y.: Vatex: a large-scale, high-quality multilingual dataset for video-and-language research. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4581\u20134591 (2019)","DOI":"10.1109\/ICCV.2019.00468"},{"issue":"4","key":"30_CR29","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., Simoncelli, E.P.: Image quality assessment: from error visibility to structural similarity. IEEE Trans. Image Process. 13(4), 600\u2013612 (2004)","journal-title":"IEEE Trans. Image Process."},{"key":"30_CR30","doi-asserted-by":"crossref","unstructured":"Wu, Z., Kong, L., Bi, W., Li, X., Kao, B.: Good for misconceived reasons: an empirical revisiting on the need for visual context in multimodal machine translation. arXiv preprint arXiv:2105.14462 (2021)","DOI":"10.18653\/v1\/2021.acl-long.480"},{"key":"30_CR31","first-page":"388","volume":"30","author":"Z Yang","year":"2022","unstructured":"Yang, Z., Hirasawa, T., Komachi, M., Okazaki, N.: Why videos do not guide translations in video-guided machine translation? An empirical evaluation of video-guided machine translation dataset. J. Inf. Process. 30, 388\u2013396 (2022)","journal-title":"J. Inf. Process."},{"key":"30_CR32","doi-asserted-by":"crossref","unstructured":"Yao, S., Wan, X.: Multimodal transformer for multimodal machine translation. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 4346\u20134350 (2020)","DOI":"10.18653\/v1\/2020.acl-main.400"},{"key":"30_CR33","doi-asserted-by":"crossref","unstructured":"Yin, Y., et al.: A novel graph-based multi-modal fusion encoder for neural machine translation. arXiv preprint arXiv:2007.08742 (2020)","DOI":"10.18653\/v1\/2020.acl-main.273"},{"key":"30_CR34","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Sun, Z., Cheng, S., Huang, L., Wu, L., Wang, M.: Beyond triplet: leveraging the most data for multimodal machine translation. arXiv preprint arXiv:2212.10313 (2022)","DOI":"10.18653\/v1\/2023.findings-acl.168"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-97141-9_30","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T04:59:46Z","timestamp":1777525186000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-97141-9_30"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,1]]},"ISBN":["9783031971402","9783031971419"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-97141-9_30","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,1]]},"assertion":[{"value":"1 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLDB","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Applications of Natural Language to Information Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kanazawa","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nldb2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/nldb2025.github.io\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}