{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T06:42:22Z","timestamp":1774680142913,"version":"3.50.1"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031959172","type":"print"},{"value":"9783031959189","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-95918-9_25","type":"book-chapter","created":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T13:30:54Z","timestamp":1750512654000},"page":"356-369","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Food State Recognition from\u00a0Recipes Using Multimodal Model for\u00a0Task Monitoring in\u00a0Autonomous Cooking Robots"],"prefix":"10.1007","author":[{"given":"Rina","family":"Tagami","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hiroki","family":"Kobayashi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuichi","family":"Akizuki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manabu","family":"Hashimoto","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,16]]},"reference":[{"key":"25_CR1","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763, PMLR (2021)"},{"key":"25_CR2","doi-asserted-by":"crossref","unstructured":"Kanazawa, N., Kawaharazuka, K., Obinata, Y., Okada, K., Inaba, M.: Recognition of heat-induced food state changes by time-series use of vision-language model for cooking robot. In: International Conference on Intelligent Autonomous Systems, pp. 547\u2013560, Springer, Cham (2023)","DOI":"10.1007\/978-3-031-44851-5_42"},{"key":"25_CR3","doi-asserted-by":"crossref","unstructured":"Kawaharazuka, K., Kanazawa, N., Obinata, Y., Okada, K., Inaba, M.: Continuous object state recognition for cooking robots using pre-trained vision-language models and black-box optimization. IEEE Robot. Autom. Lett. (2024)","DOI":"10.1109\/LRA.2024.3375257"},{"issue":"18","key":"25_CR4","doi-asserted-by":"publisher","first-page":"1318","DOI":"10.1080\/01691864.2024.2407136","volume":"38","author":"N Kanazawa","year":"2024","unstructured":"Kanazawa, N., Kawaharazuka, K., Obinata, Y., Okada, K., Inaba, M.: Real-world cooking robot system from recipes based on food state recognition using foundation models and PDDL. Adv. Robot. 38(18), 1318\u20131334 (2024)","journal-title":"Adv. Robot."},{"key":"25_CR5","unstructured":"Dosovitskiy, A.: An image is worth 16x16 words: transformers for image recognition at scale, arXiv preprint arXiv:2010.11929 (2020)"},{"key":"25_CR6","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916, PMLR (2021)"},{"key":"25_CR7","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742 PMLR (2023)"},{"key":"25_CR8","doi-asserted-by":"crossref","unstructured":"Min, W., Liu, L., Wang, Z., Luo, Z., Wei, X., Wei, X., Jiang, S.: food-500: a dataset for large-scale food recognition via stacked global-local attention network. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 393\u2013401 (2020)","DOI":"10.1145\/3394171.3414031"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"Zou, Z., Zhu, X., Zhu, Q., Liu, Y., Zhu, L.: CREAMY: cross-modal recipe retrieval by avoiding matching imperfectly. IEEE Access (2024)","DOI":"10.1109\/ACCESS.2024.3370158"},{"key":"25_CR10","doi-asserted-by":"crossref","unstructured":"Xu, F.F., Ji, L., Shi, B., Du, J., Neubig, G.B.Y., Duan, N.: A benchmark for structured procedural knowledge extraction from cooking videos, arXiv preprint arXiv:2005.00706 (2020)","DOI":"10.18653\/v1\/2020.nlpbt-1.4"},{"key":"25_CR11","doi-asserted-by":"crossref","unstructured":"Papadopoulos, D.P., Mora, E., Chepurko, N., Huang, K.W., Ofli, F., Torralba, A.: Learning program representations for food images and cooking recipes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16559\u201316569 (2022)","DOI":"10.1109\/CVPR52688.2022.01606"},{"key":"25_CR12","doi-asserted-by":"crossref","unstructured":"Fu, H., Wu, R., Liu, C., Sun, J.: Mcen: bridging cross-modal gap between cooking recipes and dish images with latent variable model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14570\u201314580 (2020)","DOI":"10.1109\/CVPR42600.2020.01458"},{"key":"25_CR13","doi-asserted-by":"crossref","unstructured":"Li, J., et al.: Hybrid fusion with intra-and cross-modality attention for image-recipe retrieval. In: Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 244\u2013254 (2021)","DOI":"10.1145\/3404835.3462965"},{"key":"25_CR14","doi-asserted-by":"crossref","unstructured":"Shukor, M., Couairon, G., Grechka, A., Cord, M.: Transformer decoders with multimodal regularization for cross-modal food retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4567\u20134578 (2022)","DOI":"10.1109\/CVPRW56347.2022.00503"},{"key":"25_CR15","doi-asserted-by":"publisher","first-page":"2515","DOI":"10.1109\/TMM.2021.3083109","volume":"24","author":"H Wang","year":"2021","unstructured":"Wang, H., et al.: Cross-modal food retrieval: learning a joint embedding of food images and recipes with semantic consistency and attention mechanism. IEEE Trans. Multimedia 24, 2515\u20132525 (2021)","journal-title":"IEEE Trans. Multimedia"},{"key":"25_CR16","doi-asserted-by":"crossref","unstructured":"Wahed, M., Zhou, X., Yu, T., Lourentzou, I.: Fine-grained alignment for cross-modal recipe retrieval. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5584\u20135593 (2024)","DOI":"10.1109\/WACV57701.2024.00549"},{"key":"25_CR17","doi-asserted-by":"crossref","unstructured":"Salvador, A., Gundogdu, E., Bazzani, L., Donoser, M.: Revamping cross-modal recipe retrieval with hierarchical transformers and self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15475\u201315484 (2021)","DOI":"10.1109\/CVPR46437.2021.01522"},{"key":"25_CR18","first-page":"35946","volume":"35","author":"C Feichtenhofer","year":"2022","unstructured":"Feichtenhofer, C., Li, Y., He, K.: Masked autoencoders as spatiotemporal learners. Adv. Neural. Inf. Process. Syst. 35, 35946\u201335958 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Woo, S., et al.: Convnext v2: co-designing and scaling convnets with masked autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16133\u201316142 (2023)","DOI":"10.1109\/CVPR52729.2023.01548"},{"key":"25_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., Fan, H., Hu, R., Feichtenhofer, C., He, K.: Scaling language-image pre-training via masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23390\u201323400 (2023)","DOI":"10.1109\/CVPR52729.2023.02240"},{"key":"25_CR21","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A.G., Kirillov, A., Girdhar, R.: Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1290\u20131299 (2022)","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"25_CR22","unstructured":"Zhou, L., Louis, N., Corso, J.J.: Weakly-supervised video object grounding from text by loss weighting and object interaction, arXiv preprint arXiv:1805.02834 (2018)"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Sener, F., Yao, A.: Zero-shot anticipation for instructional activities. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 862\u2013871 (2019)","DOI":"10.1109\/ICCV.2019.00095"},{"issue":"6","key":"25_CR24","first-page":"5998","volume":"3","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., et al.: Attention is all you need. NIPS 3(6), 5998\u20136008 (2017)","journal-title":"NIPS"},{"key":"25_CR25","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (ICLR), vol.5, no.6 (2019)"}],"container-title":["Lecture Notes in Computer Science","Image Analysis"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-95918-9_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T05:16:28Z","timestamp":1774674988000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-95918-9_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031959172","9783031959189"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-95918-9_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"16 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SCIA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Scandinavian Conference on Image Analysis","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Reykjavik","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Iceland","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 June 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"scia2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/scia2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}