{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:52:41Z","timestamp":1781931161524,"version":"3.54.5"},"reference-count":45,"publisher":"Association for Natural Language Processing","issue":"2","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Journal of Natural Language Processing"],"published-print":{"date-parts":[[2026]]},"DOI":"10.5715\/jnlp.33.537","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:11:39Z","timestamp":1781475099000},"page":"537-569","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing Image Clustering with Captions"],"prefix":"10.5715","volume":"33","author":[{"given":"Yuanyuan","family":"Cai","sequence":"first","affiliation":[{"name":"Institute of Science Tokyo"},{"name":"Mazda Motor Corporation"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Satoshi","family":"Kosugi","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kotaro","family":"Funakoshi","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manabu","family":"Okumura","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"3685","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"Al-Tameemi, I. S., Feizi-Derakhshi, M.-R., Pashazadeh, S., and Asadpour, M. (2023). \u201cMulti-model Fusion Framework using Deep Learning for Visual-textual Sentiment Classification.\u201d <i>Computers, Materials &amp; Continua<\/i>, <b>76<\/b> (2), pp. 2145\u20132177.","DOI":"10.32604\/cmc.2023.040997"},{"key":"2","unstructured":"Bai, S., Chen, K., Liu, X., Wang, J., Ge, W., Song, S., Dang, K., Wang, P., Wang, S., Tang, J., et al. (2025). \u201cQwen2. 5-vl Technical Report.\u201d <i>arXiv preprint arXiv:2502.13923<\/i>."},{"key":"3","doi-asserted-by":"crossref","unstructured":"Bakkali, S., Ming, Z., Coustaty, M., and Rusi\u00f1ol, M. (2020). \u201cVisual and Textual Deep Feature Fusion for Document Image Classification.\u201d In <i>Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops<\/i>, pp. 562\u2013563.","DOI":"10.1109\/CVPRW50498.2020.00289"},{"key":"4","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., and Van Gool, L. (2014). \u201cFood-101\u2013mining Discriminative Components with Random Forests.\u201d In <i>Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part VI 13<\/i>, pp. 446\u2013461. Springer.","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"5","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al. (2020). \u201cLanguage models are Few-shot Learners.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>33<\/b>, pp. 1877\u20131901."},{"key":"6","unstructured":"Cai, Y., Kosugi, S., Funakoshi, K., and Okumura, M. (2024). \u201cEnhancing Image Clustering with Captions.\u201d In Oco, N., Dita, S. N., Borlongan, A. M., and Kim, J.-B. (Eds.), <i>Proceedings of the 38th Pacific Asia Conference on Language, Information and Computation<\/i>, pp. 246\u2013255, Tokyo, Japan. Tokyo University of Foreign Studies."},{"key":"7","doi-asserted-by":"crossref","unstructured":"Chow, W., Li, J., Yu, Q., Pan, K., Fei, H., Ge, Z., Yang, S., Tang, S., Zhang, H., and Sun, Q. (2024). \u201cUnified Generative and Discriminative Training for Multi-modal Large Language Models.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>37<\/b>, pp. 23155\u201323190.","DOI":"10.52202\/079017-0729"},{"key":"8","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., and Vedaldi, A. (2014). \u201cDescribing Textures in the Wild.\u201d In <i>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition<\/i>, pp. 3606\u20133613.","DOI":"10.1109\/CVPR.2014.461"},{"key":"9","unstructured":"Comanici, G., Bieber, E., Schaekermann, M., Pasupat, I., Sachdeva, N., Dhillon, I., Blistein, M., Ram, O., Zhang, D., Rosen, E., et al. (2025). \u201cGemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities.\u201d <i>arXiv preprint arXiv:2507.06261<\/i>."},{"key":"10","unstructured":"Davidson, I., Gourru, A., and Ravi, S. (2018). \u201cThe Cluster Description Problem-complexity Results, Formulations and Approximations.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>31<\/b>."},{"key":"11","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., and Fei-Fei, L. (2009). \u201cImagenet: A Large-Scale Hierarchical Image Database.\u201d In <i>2009 IEEE Conference on Computer Vision and Pattern Recognition<\/i>, pp. 248\u2013255. IEEE.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"12","doi-asserted-by":"crossref","unstructured":"Do, T. D., Kim, K., Park, H., and Yang, H.-J. (2020). \u201cImage and Encoded Text Fusion for Deep Multi-modal Clustering.\u201d In <i>The 9th International Conference on Smart Media and Applications<\/i>, pp. 308\u2013312.","DOI":"10.1145\/3426020.3426110"},{"key":"13","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., and Perona, P. (2005). \u201cA Bayesian Hierarchical Model for Learning Natural Scene Categories.\u201d In <i>2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201905)<\/i>, Vol. 2, pp. 524\u2013531. IEEE.","DOI":"10.1109\/CVPR.2005.16"},{"key":"14","doi-asserted-by":"crossref","unstructured":"He, X., and Peng, Y. (2019). \u201cFine-grained Visual-textual Representation Learning.\u201d <i>IEEE Transactions on Circuits and Systems for Video Technology<\/i>, <b>30<\/b> (2), pp. 520\u2013531.","DOI":"10.1109\/TCSVT.2019.2892802"},{"key":"15","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Bras, R. L., and Choi, Y. (2021). \u201cClipscore: A Reference-Free Evaluation Metric for Image Captioning.\u201d <i>arXiv preprint arXiv:2104.08718<\/i>.","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"16","doi-asserted-by":"crossref","unstructured":"Hubert, L., and Arabie, P. (1985). \u201cComparing Partitions.\u201d <i>Journal of Classification<\/i>, <b>2<\/b>, pp. 193\u2013218.","DOI":"10.1007\/BF01908075"},{"key":"17","doi-asserted-by":"crossref","unstructured":"Kuhn, H. W. (1955). \u201cThe Hungarian Method for the Assignment Problem.\u201d <i>Naval Research Logistics Quarterly<\/i>, <b>2<\/b> (1\u20132), pp. 83\u201397.","DOI":"10.1002\/nav.3800020109"},{"key":"18","unstructured":"Kwon, S., Park, J., Kim, M., Cho, J., Ryu, E. K., and Lee, K. (2024). \u201cImage Clustering Conditioned on Text Criteria.\u201d <i>International Conference on Learning Representations<\/i>."},{"key":"19","doi-asserted-by":"crossref","unstructured":"Lazebnik, S., Schmid, C., and Ponce, J. (2006). \u201cBeyond Bags of Features: Spatial Pyramid Matching for Recognizing Natural Scene Categories.\u201d In <i>2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201906)<\/i>, Vol. 2, pp. 2169\u20132178. IEEE.","DOI":"10.1109\/CVPR.2006.68"},{"key":"20","unstructured":"Li, J., Li, D., Savarese, S., and Hoi, S. (2023). \u201cBlip-2: Bootstrapping Language-image Pre-training with Frozen Image Encoders and Large Language Models.\u201d In <i>International Conference on Machine Learning<\/i>, pp. 19730\u201319742. PMLR."},{"key":"21","unstructured":"Li, J., Li, D., Xiong, C., and Hoi, S. (2022). \u201cBlip: Bootstrapping Language-image Pre-training for Unified Vision-language Understanding and Generation.\u201d In <i>International Conference on Machine Learning<\/i>, pp. 12888\u201312900. PMLR."},{"key":"22","doi-asserted-by":"crossref","unstructured":"Li, Y., Hu, P., Liu, Z., Peng, D., Zhou, J. T., and Peng, X. (2021). \u201cContrastive Clustering.\u201d In <i>Proceedings of the AAAI Conference on Artificial Intelligence<\/i>, Vol. 35, pp. 8547\u20138555.","DOI":"10.1609\/aaai.v35i10.17037"},{"key":"23","unstructured":"Li, Y., Hu, P., Peng, D., Lv, J., Fan, J., and Peng, X. (2023). \u201cImage Clustering with External Guidance.\u201d <i>arXiv preprint arXiv:2310.11989<\/i>."},{"key":"24","unstructured":"Lyons, M. J., Akamatsu, S., Kamachi, M., Gyoba, J., and Budynek, J. (1998). \u201cThe Japanese Female Facial Expression (JAFFE) Database.\u201d In <i>Proceedings of 3rd International Conference on Automatic Face and Gesture Recognition<\/i>, pp. 14\u201316."},{"key":"25","unstructured":"MacQueen, J. (1967). \u201cSome Methods for Classification and Analysis of Multivariate Observations.\u201d In <i>Proceedings of the 5th Berkeley Symposium on Mathematical Statistics and Probability<\/i>, Vol. 1, pp. 281\u2013297. Oakland, CA, USA."},{"key":"26","unstructured":"Menon, S., and Vondrick, C. (2023). \u201cVisual Classification via Description from Large Language Models.\u201d <i>International Conference on Learning Representations<\/i>."},{"key":"27","unstructured":"Mokady, R., Hertz, A., and Bermano, A. H. (2021). \u201cClipcap: Clip Prefix for Image Captioning.\u201d <i>arXiv preprint arXiv:2111.09734<\/i>."},{"key":"28","doi-asserted-by":"crossref","unstructured":"Oliva, A., and Torralba, A. (2001). \u201cModeling the Shape of the Scene: A Holistic Representation of the Spatial Envelope.\u201d <i>International Journal of Computer Vision<\/i>, <b>42<\/b>, pp. 145\u2013175.","DOI":"10.1023\/A:1011139631724"},{"key":"29","unstructured":"OpenAI (2023). \u201cChatGPT (Mar 14 version) [Large Language Model].\u201d Retrieved from https:\/\/chat.openai.com\/chat."},{"key":"30","doi-asserted-by":"crossref","unstructured":"Panickssery, A., Bowman, S., and Feng, S. (2024). \u201cLlm Evaluators Recognize and Favor Their Own Generations.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>37<\/b>, pp. 68772\u201368802.","DOI":"10.52202\/079017-2197"},{"key":"31","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., and Sutskever, I. (2021). \u201cLearning Transferable Visual Models from Natural Language Supervision.\u201d In <i>International Conference on Machine Learning<\/i>, pp. 8748\u20138763. PMLR."},{"key":"32","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W., and Liu, P. J. (2020). \u201cExploring the Limits of Transfer Learning with a Unified Text-to-text Transformer.\u201d <i>Journal of Machine Learning Research<\/i>, <b>21<\/b> (140), pp. 1\u201367."},{"key":"33","doi-asserted-by":"crossref","unstructured":"Saito, K., Kim, D., Park, K., Hashimoto, A., and Ushiku, Y. (2025). \u201cCaptionSmiths: Flexibly Controlling Language Pattern in Image Captioning.\u201d <i>arXiv preprint arXiv:2507.01409<\/i>.","DOI":"10.1109\/ICCV51701.2025.01848"},{"key":"34","doi-asserted-by":"crossref","unstructured":"Sambaturu, P., Gupta, A., Davidson, I., Ravi, S., Vullikanti, A., and Warren, A. (2020). \u201cEfficient Algorithms for Generating Provably Near-optimal Cluster Descriptors for Explainability.\u201d In <i>Proceedings of the AAAI Conference on Artificial Intelligence<\/i>, Vol. 34, pp. 1636\u20131643.","DOI":"10.1609\/aaai.v34i02.5525"},{"key":"35","unstructured":"Shen, Y., Shen, Z., Wang, M., Qin, J., Torr, P., and Shao, L. (2021). \u201cYou Never Cluster Alone.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>34<\/b>, pp. 27734\u201327746."},{"key":"36","doi-asserted-by":"crossref","unstructured":"Stephan, A., Miklautz, L., Sidak, K., Wahle, J. P., Gipp, B., Plant, C., and Roth, B. (2024). \u201cText-Guided Image Clustering.\u201d In Graham, Y., and Purver, M. (Eds.), <i>Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers)<\/i>, pp. 2960\u20132976, St. Julian\u2019s, Malta. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.eacl-long.180"},{"key":"37","doi-asserted-by":"crossref","unstructured":"Tembhurne, J. V., and Diwan, T. (2021). \u201cSentiment Analysis in Textual, Visual and Multimodal Inputs Using Recurrent Neural Networks.\u201d <i>Multimedia Tools and Applications<\/i>, <b>80<\/b> (5), pp. 6871\u20136910.","DOI":"10.1007\/s11042-020-10037-x"},{"key":"38","unstructured":"Vinh, N. X., Epps, J., and Bailey, J. (2010). \u201cInformation Theoretic Measures for Clusterings Comparison: Variants, Properties, Normalization and Correction for Chance.\u201d <i>Journal of Machine Learning Research<\/i>, <b>11<\/b> (95), pp. 2837\u20132854."},{"key":"39","doi-asserted-by":"crossref","unstructured":"Wolfe, R., and Caliskan, A. (2022). \u201cContrastive Visual Semantic Pretraining Magnifies the Semantics of Natural Language Representations.\u201d <i>arXiv preprint arXiv:2203.07511<\/i>.","DOI":"10.18653\/v1\/2022.acl-long.217"},{"key":"40","doi-asserted-by":"crossref","unstructured":"Xiao, H., Zhang, F., Shen, Z., Wu, K., and Zhang, J. (2021). \u201cClassification of Weather Phenomenon from Images by using Deep Convolutional Neural Network.\u201d <i>Earth and Space Science<\/i>, <b>8<\/b> (5), p. e2020EA001604.","DOI":"10.1029\/2020EA001604"},{"key":"41","doi-asserted-by":"crossref","unstructured":"Yang, Y., Xu, D., Nie, F., Yan, S., and Zhuang, Y. (2010). \u201cImage Clustering using Local Discriminant Models and Global Integration.\u201d <i>IEEE Transactions on Image Processing<\/i>, <b>19<\/b> (10), pp. 2761\u20132773.","DOI":"10.1109\/TIP.2010.2049235"},{"key":"42","doi-asserted-by":"crossref","unstructured":"Yao, B., Jiang, X., Khosla, A., Lin, A. L., Guibas, L., and Fei-Fei, L. (2011). \u201cHuman Action Recognition by Learning Bases of Action Attributes and Parts.\u201d In <i>2011 International Conference on Computer Vision<\/i>, pp. 1331\u20131338. IEEE.","DOI":"10.1109\/ICCV.2011.6126386"},{"key":"43","unstructured":"Zhang, H., and Davidson, I. (2021). \u201cDeep Descriptive Clustering.\u201d <i>arXiv preprint arXiv:2105.11549<\/i>."},{"key":"44","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Gao, T., and Guo, N. (2023). \u201cTsvfn: Two-stage Visual Fusion Network for Multimodal Relation Extraction.\u201d <i>Information Processing &amp; Management<\/i>, <b>60<\/b> (3), p. 103264.","DOI":"10.1016\/j.ipm.2023.103264"},{"key":"45","doi-asserted-by":"crossref","unstructured":"Zhong, H., Wu, J., Chen, C., Huang, J., Deng, M., Nie, L., Lin, Z., and Hua, X.-S. (2021). \u201cGraph Contrastive Clustering.\u201d In <i>Proceedings of the IEEE\/CVF International Conference on Computer Vision<\/i>, pp. 9224\u20139233.","DOI":"10.1109\/ICCV48922.2021.00909"}],"container-title":["Journal of Natural Language Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_537\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:43:51Z","timestamp":1781930631000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_537\/_article\/-char\/ja\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":45,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.5715\/jnlp.33.537","relation":{},"ISSN":["1340-7619","2185-8314"],"issn-type":[{"value":"1340-7619","type":"print"},{"value":"2185-8314","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}