{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T18:08:05Z","timestamp":1743012485361,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":20,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819620708"},{"type":"electronic","value":"9789819620715"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-2071-5_31","type":"book-chapter","created":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T15:34:52Z","timestamp":1735745692000},"page":"428-441","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Quantifying Image-Adjective Associations by\u00a0Leveraging Large-Scale Pretrained Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2453-4560","authenticated-orcid":false,"given":"Chihaya","family":"Matsuhira","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9193-5973","authenticated-orcid":false,"given":"Marc A.","family":"Kastner","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3041-4330","authenticated-orcid":false,"given":"Takahiro","family":"Komamizu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6290-9680","authenticated-orcid":false,"given":"Takatsugu","family":"Hirayama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3942-9296","authenticated-orcid":false,"given":"Ichiro","family":"Ide","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,2]]},"reference":[{"key":"31_CR1","first-page":"78347","volume":"36","author":"M Alper","year":"2023","unstructured":"Alper, M., Averbuch-Elor, H.: Kiki or Bouba? sound symbolism in vision-and-language models. Adv. Neural Inform. Process. Syst. 36, 78347\u201378359 (2023)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"31_CR2","unstructured":"Bird, S., Klein, E., Loper, E.: Natural language processing with Python: analyzing text with the Natural Language Toolkit. O\u2019Reilly Media Inc, Sebastopol, CA, USA (2009)"},{"key":"31_CR3","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural Inform. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"31_CR4","unstructured":"Computer Vision and Learning Research Group at Ludwig Maximilian University of Munich: Stable diffusion (2022). https:\/\/github.com\/CompVis\/stable-diffusion\/ (Accessed 22 July 2024)"},{"key":"31_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1007\/11744078_23","volume-title":"Computer Vision \u2013 ECCV 2006","author":"R Datta","year":"2006","unstructured":"Datta, R., Joshi, D., Li, J., Wang, J.Z.: Studying aesthetics in photographic images using a computational approach. In: Leonardis, A., Bischof, H., Pinz, A. (eds.) ECCV 2006. LNCS, vol. 3953, pp. 288\u2013301. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11744078_23"},{"key":"31_CR6","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186 (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"issue":"976235","key":"31_CR7","doi-asserted-by":"publisher","first-page":"11p","DOI":"10.3389\/frai.2022.976235","volume":"5","author":"S Hentschel","year":"2022","unstructured":"Hentschel, S., Kobs, K., Hotho, A.: CLIP knows image aesthetics. Front. Artif. Intell. 5(976235), 11p (2022). https:\/\/doi.org\/10.3389\/frai.2022.976235","journal-title":"Front. Artif. Intell."},{"key":"31_CR8","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: Proceedings of the 38th International Conference on Machine Learning, vol.\u00a0139, pp. 4904\u20134916 (2021)"},{"key":"31_CR9","doi-asserted-by":"publisher","unstructured":"Jiang, A.Q., et al.: Mistral 7B. Comput. Res. Reposit., arXiv Preprints, arXiv:2310.06825 (2023). https:\/\/doi.org\/10.48550\/arxiv.2310.06825","DOI":"10.48550\/arxiv.2310.06825"},{"key":"31_CR10","doi-asserted-by":"publisher","first-page":"183","DOI":"10.1162\/tacl_a_00132","volume":"3","author":"A Lazaridou","year":"2015","unstructured":"Lazaridou, A., Dinu, G., Liska, A., Baroni, M.: From visual attributes to adjectives through decompositional distributional semantics. Trans. Assoc. Comput. Linguist. 3, 183\u2013196 (2015). https:\/\/doi.org\/10.1162\/tacl_a_00132","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"31_CR11","unstructured":"Mikolov, T., Chen, K., Corrado, G.S., Dean, J.: Efficient estimation of word representations in vector space. Comput. Res. Reposit., arXiv Preprints, arXiv:1301.3781 (2013)"},{"key":"31_CR12","doi-asserted-by":"publisher","unstructured":"Murray, N., Marchesotti, L., Perronnin, F.: AVA: a large-scale database for aesthetic visual analysis. In: Proceedings of 2012 IEEE Conference on Computer Vision and Pattern Recognition, pp. 2408\u20132415 (2012). https:\/\/doi.org\/10.1109\/CVPR.2012.6247954","DOI":"10.1109\/CVPR.2012.6247954"},{"key":"31_CR13","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763 (2021)"},{"key":"31_CR14","doi-asserted-by":"publisher","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01042","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"31_CR15","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-642-35749-7_1","volume-title":"Trends and Topics in Computer Vision","author":"O Russakovsky","year":"2012","unstructured":"Russakovsky, O., Fei-Fei, L.: Attribute learning in large-scale datasets. In: Kutulakos, K.N. (ed.) ECCV 2010. LNCS, vol. 6553, pp. 1\u201314. Springer, Heidelberg (2012). https:\/\/doi.org\/10.1007\/978-3-642-35749-7_1"},{"key":"31_CR16","doi-asserted-by":"publisher","unstructured":"Shimoda, W., Yanai, K.: A visual analysis on recognizability and discriminability of onomatopoeia words with DCNN features. In: Proceedings of 2015 IEEE International Conference on Multimedia and Expo, 6 p. (2015). https:\/\/doi.org\/10.1109\/ICME.2015.7177453","DOI":"10.1109\/ICME.2015.7177453"},{"key":"31_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1007\/978-3-540-30541-5_25","volume-title":"Advances in Multimedia Information Processing - PCM 2004","author":"H Tong","year":"2004","unstructured":"Tong, H., Li, M., Zhang, H.-J., He, J., Zhang, C.: Classification of digital photos taken by photographers or home users. In: Aizawa, K., Nakamura, Y., Satoh, S. (eds.) PCM 2004. LNCS, vol. 3331, pp. 198\u2013205. Springer, Heidelberg (2004). https:\/\/doi.org\/10.1007\/978-3-540-30541-5_25"},{"key":"31_CR18","unstructured":"Touvron, H., et al.: LLaMA: open and efficient foundation language models. Comput. Res. Reposit., arXiv Preprints, arXiv:2302.13971 (2023)"},{"key":"31_CR19","doi-asserted-by":"publisher","unstructured":"Wang, J., Chan, K.C., Loy, C.C.: Exploring CLIP for assessing the look and feel of images. In: Proceedings of AAAI Conference on Artificial Intelligence, vol. 37(2), pp. 2555\u20132563 (2023). https:\/\/doi.org\/10.1609\/aaai.v37i2.25353","DOI":"10.1609\/aaai.v37i2.25353"},{"issue":"4","key":"31_CR20","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3197517.3201355","volume":"37","author":"N Zhao","year":"2018","unstructured":"Zhao, N., Cao, Y., Lau, R.W.: What characterizes personalities of graphic designs? ACM Trans. Graph. 37(4), 1\u201315 (2018). https:\/\/doi.org\/10.1145\/3197517.3201355","journal-title":"ACM Trans. Graph."}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-2071-5_31","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T16:04:09Z","timestamp":1735747449000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-2071-5_31"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819620708","9789819620715"],"references-count":20,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-2071-5_31","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"2 January 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Nara","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 January 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 January 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/mmm2025.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}