{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:52:44Z","timestamp":1781931164094,"version":"3.54.5"},"reference-count":69,"publisher":"Association for Natural Language Processing","issue":"2","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Journal of Natural Language Processing"],"published-print":{"date-parts":[[2026]]},"DOI":"10.5715\/jnlp.33.591","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:11:38Z","timestamp":1781475098000},"page":"591-629","source":"Crossref","is-referenced-by-count":0,"title":["Ruri: Japanese General Text Embeddings","Ruri\uff1a\u65e5\u672c\u8a9e\u306b\u7279\u5316\u3057\u305f\u6c4e\u7528\u30c6\u30ad\u30b9\u30c8\u57cb\u3081\u8fbc\u307f\u30e2\u30c7\u30eb"],"prefix":"10.5715","volume":"33","author":[{"given":"Hayato","family":"Tsukagoshi","sequence":"first","affiliation":[{"name":"Nagoya University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ryohei","family":"Sasano","sequence":"additional","affiliation":[{"name":"Nagoya University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"3685","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"Ait-Saada, M., and Nadif, M. (2023). \u201cIs Anisotropy Truly Harmful? A Case Study on Text Clustering.\u201d  In <i>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (ACL)<\/i>, pp. 1194\u20131203.","DOI":"10.18653\/v1\/2023.acl-short.103"},{"key":"2","unstructured":"Bonifacio, L., Jeronymo, V., Abonizio, H. Q., Campiotti, I., Fadaee, M., Lotufo, R., and Nogueira, R. (2021). \u201cmMARCO: A Multilingual Version of the MS MARCO Passage Ranking Dataset.\u201d  <i>arXiv preprint arXiv:2108.13897<\/i>."},{"key":"3","doi-asserted-by":"crossref","unstructured":"Bowman, S. R., Angeli, G., Potts, C., and Manning, C. D. (2015). \u201cA large annotated corpus for learning natural language inference.\u201d  In <i>Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing (EMNLP)<\/i>, pp. 632\u2013642.","DOI":"10.18653\/v1\/D15-1075"},{"key":"4","unstructured":"Clavi\u00e9, B. (2023). \u201cJaColBERT and Hard Negatives, Towards Better Japanese-First Embeddings for Retrieval: Early Technical Report.\u201d  <i>arXiv preprint arXiv:2312.16144<\/i>."},{"key":"5","doi-asserted-by":"crossref","unstructured":"Clavi\u00e9, B. (2024). \u201cJaColBERTv2.5: Optimising Multi-Vector Retrievers to Create State-of-the-Art Japanese Retrievers with Constrained Resources.\u201d  <i>arXiv preprint arXiv:2407.20750<\/i>.","DOI":"10.5715\/jnlp.32.176"},{"key":"6","doi-asserted-by":"crossref","unstructured":"Cormack, G. V., Clarke, C. L. A., and Buettcher, S. (2009). \u201cReciprocal Rank Fusion Outperforms Condorcet and Individual Rank Learning Methods.\u201d  In <i>Proceedings of the 32nd International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)<\/i>, pp. 758\u2013759.","DOI":"10.1145\/1571941.1572114"},{"key":"7","unstructured":"Dao, T. (2023). \u201cFlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning.\u201d  <i>arXiv preprint arXiv:2307.08691<\/i>."},{"key":"8","doi-asserted-by":"crossref","unstructured":"de Souza P. Moreira, G., Osmulski, R., Xu, M., Ak, R., Schifferer, B., and Oldridge, E. (2024). \u201cNV-Retriever: Improving Text Embedding Models with Effective Hard-negative Mining.\u201d  <i>arXiv preprint arXiv:2407.15831<\/i>.","DOI":"10.1145\/3746252.3761254"},{"key":"9","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.-W., Lee, K., and Toutanova, K. (2019). \u201cBERT: Pre-training of Deep Bidirectional Transformers for Language Understanding.\u201d  In <i>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL)<\/i>, pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10","doi-asserted-by":"crossref","unstructured":"Faruqui, M., Pavlick, E., Tenney, I., and Das, D. (2018). \u201cWikiAtomicEdits: A Multilingual Corpus of Wikipedia Edits for Modeling Language and Discourse.\u201d  In <i>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing (EMNLP 2018)<\/i>, pp. 305\u2013315.","DOI":"10.18653\/v1\/D18-1028"},{"key":"11","doi-asserted-by":"crossref","unstructured":"Formal, T., Piwowarski, B., and Clinchant, S. (2021). \u201cSPLADE: Sparse Lexical and Expansion Model for First Stage Ranking.\u201d  In <i>Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)<\/i>, pp. 2288\u20132292.","DOI":"10.1145\/3404835.3463098"},{"key":"12","unstructured":"G\u00fcnther, M., Ong, J., Mohr, I., Abdessalem, A., Abel, T., Akram, M. K., Guzman, S., Mastrapas, G., Sturua, S., Wang, B., Werk, M., Wang, N., and Xiao, H. (2024). \u201cJina Embeddings 2: 8192-Token General-Purpose Text Embeddings for Long Documents.\u201d  <i>arXiv preprint arXiv:2310.19923<\/i>."},{"key":"13","unstructured":"Ilharco, G., Ribeiro, M. T., Wortsman, M., Schmidt, L., Hajishirzi, H., and Farhadi, A. (2023). \u201cEditing Models with Task Arithmetic.\u201d  In <i>The 11th International Conference on Learning Representations (ICLR)<\/i>."},{"key":"14","doi-asserted-by":"crossref","unstructured":"Isahara, H., Bond, F., Uchimoto, K., Utiyama, M., and Kanzaki, K. (2008). \u201cDevelopment of the Japanese WordNet.\u201d  In <i>Proceedings of the 6th International Conference on Language Resources and Evaluation (LREC)<\/i>, pp. 2420\u20132423.","DOI":"10.63317\/3b5mysdrfy59"},{"key":"15","unstructured":"Ishigami, R. (2024). \u201ccyberagent\/calm3-22b-chat.\u201d  https:\/\/huggingface.co\/cyberagent\/calm3-22b-chat."},{"key":"16","unstructured":"Ishigami, R. (2025). \u201cDeepSeek-R1-Distill-Qwen-32B-Japanese.\u201d  https:\/\/huggingface.co\/cyberagent\/DeepSeek-R1-Distill-Qwen-32B-Japanese."},{"key":"17","doi-asserted-by":"crossref","unstructured":"J\u00e4rvelin, K., and Kek\u00e4l\u00e4inen, J. (2002). \u201cCumulated Gain-based Evaluation of IR Techniques.\u201d  <i>ACM Transactions on Information Systems<\/i>, <b>20<\/b>, pp. 422\u2013446.","DOI":"10.1145\/582415.582418"},{"key":"18","doi-asserted-by":"crossref","unstructured":"Katsuta, A., and Yamamoto, K. (2018). \u201cCrowdsourced Corpus of Sentence Simplification with Core Vocabulary.\u201d  In <i>Proceedings of the 11th International Conference on Language Resources and Evaluation (LREC)<\/i>, pp. 461\u2013466.","DOI":"10.63317\/2a5ax2twwiob"},{"key":"19","unstructured":"Kudo, T., Yamamoto, K., and Matsumoto, Y. (2004). \u201cApplying Conditional Random Fields to Japanese Morphological Analysis.\u201d  In <i>Proceedings of the 2004 Conference on Empirical Methods in Natural Language Processing (EMNLP)<\/i>, pp. 230\u2013237. Association for Computational Linguistics."},{"key":"20","doi-asserted-by":"crossref","unstructured":"Kurihara, K., Kawahara, D., and Shibata, T. (2022). \u201cJGLUE: Japanese General Language Understanding Evaluation.\u201d  In <i>Proceedings of the 13th Language Resources and Evaluation Conference (LREC)<\/i>, pp. 2957\u20132966.","DOI":"10.63317\/4t8apzxjf2bw"},{"key":"21","unstructured":"Lee, C., Roy, R., Xu, M., Raiman, J., Shoeybi, M., Catanzaro, B., and Ping, W. (2024a). \u201cNV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models.\u201d  <i>arXiv preprint arXiv:2405.17428<\/i>."},{"key":"22","unstructured":"Lee, J., Dai, Z., Ren, X., Chen, B., Cer, D., Cole, J. R., Hui, K., Boratko, M., Kapadia, R., Ding, W., Luan, Y., Duddu, S. M. K., Abrego, G. H., Shi, W., Gupta, N., Kusupati, A., Jain, P., Jonnalagadda, S. R., Chang, M.-W., and Naim, I. (2024b). \u201cGecko: Versatile Text Embeddings Distilled from Large Language Models.\u201d  <i>arXiv preprint arXiv:2403.20327<\/i>."},{"key":"23","unstructured":"Li, S., Ohagi, M., and Ri, R. (2024). \u201cJMTEB: Japanese Massive Text Embedding Benchmark.\u201d  https:\/\/huggingface.co\/datasets\/sbintuitions\/JMTEB%7D%7D."},{"key":"24","unstructured":"Li, Z., Zhang, X., Zhang, Y., Long, D., Xie, P., and Zhang, M. (2023). \u201cTowards General Text Embeddings with Multi-stage Contrastive Learning.\u201d  <i>arXiv preprint arXiv:2308.03281<\/i>."},{"key":"25","unstructured":"LLM-jp (2024). \u201cLLM-jp: A Cross-organizational Project for the Research and Development of Fully Open Japanese LLMs.\u201d  <i>arXiv preprint arXiv:2407.03963<\/i>."},{"key":"26","doi-asserted-by":"crossref","unstructured":"Longpre, S., Lu, Y., and Daiber, J. (2021). \u201cMKQA: A Linguistically Diverse Benchmark for Multilingual Open Domain Question Answering.\u201d  <i>Transactions of the Association for Computational Linguistics (TACL)<\/i>, pp. 1389\u20131406.","DOI":"10.1162\/tacl_a_00433"},{"key":"27","doi-asserted-by":"crossref","unstructured":"Maruyama, T., and Yamamoto, K. (2018). \u201cSimplified Corpus with Core Vocabulary.\u201d  In <i>Proceedings of the 11th International Conference on Language Resources and Evaluation (LREC)<\/i>, pp. 1153\u20131160.","DOI":"10.63317\/2fm3o2dm56ux"},{"key":"28","doi-asserted-by":"crossref","unstructured":"Muennighoff, N., Tazi, N., Magne, L., and Reimers, N. (2023). \u201cMTEB: Massive Text Embedding Benchmark.\u201d  In <i>Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics (EACL)<\/i>, pp. 2014\u20132037.","DOI":"10.18653\/v1\/2023.eacl-main.148"},{"key":"29","unstructured":"Nguyen, T., Rosenberg, M., Song, X., Gao, J., Tiwary, S., Majumder, R., and Deng, L. (2016). \u201cMS MARCO: A Human Generated MAchine Reading COmprehension Dataset.\u201d  <i>arXiv preprint arXiv:1611.09268<\/i>."},{"key":"30","doi-asserted-by":"crossref","unstructured":"Nishikawa, S., Ri, R., Yamada, I., Tsuruoka, Y., and Echizen, I. (2022). \u201cEASE: Entity-Aware Contrastive Learning of Sentence Embedding.\u201d  In <i>Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL)<\/i>, pp. 3870\u20133885.","DOI":"10.18653\/v1\/2022.naacl-main.284"},{"key":"31","unstructured":"Nussbaum, Z., Morris, J. X., Duderstadt, B., and Mulyar, A. (2024). \u201cNomic Embed: Training a Reproducible Long Context Text Embedder.\u201d  <i>arXiv preprint arXiv:2402.01613<\/i>."},{"key":"32","unstructured":"Nvidia (2024). \u201cNemotron-4 340B Technical Report.\u201d  <i>arXiv preprint arXiv:2406.11704<\/i>."},{"key":"33","unstructured":"Phi-3 Team (2024). \u201cPhi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone.\u201d  <i>arXiv preprint arXiv:2404.14219<\/i>."},{"key":"34","unstructured":"Qwen Team (2025). \u201cQwen2.5 Technical Report.\u201d  <i>arXiv preprint arXiv:2412.15115<\/i>."},{"key":"35","doi-asserted-by":"crossref","unstructured":"Rasley, J., Rajbhandari, S., Ruwase, O., and He, Y. (2020). \u201cDeepSpeed: System Optimizations Enable Training Deep Learning Models with Over 100 Billion Parameters.\u201d  In <i>Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery &amp; Data Mining<\/i>, pp. 3505\u20133506.","DOI":"10.1145\/3394486.3406703"},{"key":"36","doi-asserted-by":"crossref","unstructured":"Sato, S., Tsukagoshi, H., Sasano, R., and Takeda, K. (2024). \u201cImproving Sentence Embeddings with Automatic Generation of Training Data Using Few-shot Examples.\u201d  In <i>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (ACL SRW)<\/i>, pp. 378\u2013389.","DOI":"10.18653\/v1\/2024.acl-srw.43"},{"key":"37","unstructured":"\u4f50\u85e4\u654f\u7d00\uff0c\u6a4b\u672c\u6cf0\u4e00\uff0c\u5965\u6751\u5b66 (2017). \u5358\u8a9e\u5206\u304b\u3061\u66f8\u304d\u8f9e\u66f8mecab-ipadic-NEologd\u306e\u5b9f\u88c5\u3068\u60c5\u5831\u691c\u7d22\u306b\u304a\u3051\u308b\u52b9\u679c\u7684\u306a\u4f7f\u7528\u65b9\u6cd5\u306e\u691c\u8a0e.  \u8a00\u8a9e\u51e6\u7406\u5b66\u4f1a \u7b2c23\u56de\u5e74\u6b21\u5927\u4f1a, pp. 875\u2013878. [T. Sato et al. (2017). Tango Wakachigaki Jisho mecab-ipajic-NEologd no Jisso to Joho Kensaku ni Okeru Kokateki na Shiyohoho no Kento. Proceedings of the 23th Annual Meeting of the Association for Natural Language Processing, pp. 875\u2013878.]."},{"key":"38","unstructured":"So, B., Byun, K., Kang, K., and Cho, S. (2022). \u201cJaQuAD: Japanese Question Answering Dataset for Machine Reading Comprehension.\u201d  <i>arXiv preprint arXiv:2202.01764<\/i>."},{"key":"39","doi-asserted-by":"crossref","unstructured":"Su, H., Shi, W., Kasai, J., Wang, Y., Hu, Y., Ostendorf, M., Yih, W.-t., Smith, N. A., Zettlemoyer, L., and Yu, T. (2023a). \u201cOne Embedder, Any Task: Instruction-Finetuned Text Embeddings.\u201d  In <i>Findings of the Association for Computational Linguistics: ACL 2023<\/i>, pp. 1102\u20131121.","DOI":"10.18653\/v1\/2023.findings-acl.71"},{"key":"40","doi-asserted-by":"crossref","unstructured":"Su, J., Lu, Y., Pan, S., Murtadha, A., Wen, B., and Liu, Y. (2023b). \u201cRoFormer: Enhanced Transformer with Rotary Position Embedding.\u201d  <i>arXiv preprint arXiv:2104.09864<\/i>.","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"41","unstructured":"\u9234\u6728\u6b63\u654f\uff0c\u9234\u6728\u6f64\uff0c\u677e\u7530\u8015\u53f2\uff0c\u897f\u7530\u4eac\u4ecb\uff0c\u4e95\u4e4b\u4e0a\u76f4\u4e5f (2020). JAQKET: \u30af\u30a4\u30ba\u3092\u984c\u6750\u306b\u3057\u305f\u65e5\u672c\u8a9eQA\u30c7\u30fc\u30bf\u30bb\u30c3\u30c8\u306e\u69cb\u7bc9.  \u8a00\u8a9e\u51e6\u7406\u5b66\u4f1a\u7b2c26\u56de\u5e74\u6b21\u5927\u4f1a, pp. 237\u2013240. [M. Suzuki et al. (2020). Kuizu wo Daizai nishita Nihongo QA Detasetto no Kochiku. Proceedings of the 26th Annual Meeting for the Association for Natural Language Processing, pp. 237\u2013240.]."},{"key":"42","doi-asserted-by":"crossref","unstructured":"Takaoka, K., Hisamoto, S., Kawahara, N., Sakamoto, M., Uchida, Y., and Matsumoto, Y. (2018). \u201cSudachi: A Japanese Tokenizer for Business.\u201d  In <i>Proceedings of the 11th International Conference on Language Resources and Evaluation (LREC)<\/i>, pp. 2246\u20132249.","DOI":"10.63317\/2hsnuu4nxsnq"},{"key":"43","doi-asserted-by":"crossref","unstructured":"Takeshita, S., Takeshita, Y., Ruffinelli, D., and Ponzetto, S. P. (2025). \u201cRandomly Removing 50% of Dimensions in Text Embeddings has Minimal Impact on Retrieval and Classification Tasks.\u201d  In <i>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing (EMNLP)<\/i>, pp. 27693\u201327714.","DOI":"10.18653\/v1\/2025.emnlp-main.1410"},{"key":"44","doi-asserted-by":"crossref","unstructured":"Tanaka, Y., Murawaki, Y., Kawahara, D., and Kurohashi, S. (2020). \u201cBuilding a Japanese Typo Dataset from Wikipedia\u2018s Revision History.\u201d  In <i>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics: Student Research Workshop (ACL SRW)<\/i>, pp. 230\u2013236.","DOI":"10.18653\/v1\/2020.acl-srw.31"},{"key":"45","unstructured":"Tateno, Y. (2024a). JaCWIR: Japanese Casual Web IR - \u65e5\u672c\u8a9e\u60c5\u5831\u691c\u7d22\u8a55\u4fa1\u306e\u305f\u3081\u306e\u5c0f\u898f\u6a21\u3067\u30ab\u30b8\u30e5\u30a2\u30eb\u306aWeb\u30bf\u30a4\u30c8\u30eb\u3068\u6982\u8981\u306e\u30c7\u30fc\u30bf\u30bb\u30c3\u30c8.  https:\/\/huggingface.co\/datasets\/hotchpotch\/JaCWIR."},{"key":"46","unstructured":"Tateno, Y. (2024b). JQaRA: Japanese Question Answering with Retrieval Augmentation - \u691c\u7d22\u62e1\u5f35(RAG)\u8a55\u4fa1\u306e\u305f\u3081\u306e\u65e5\u672c\u8a9eQ&amp;A\u30c7\u30fc\u30bf\u30bb\u30c3\u30c8.  https:\/\/huggingface.co\/datasets\/hotchpotch\/JQaRA."},{"key":"47","unstructured":"Tateno, Y. (2025). \u201cFineWeb2 Edu Japanese.\u201d  https:\/\/huggingface.co\/datasets\/hotchpotch\/fineweb-2-edu-japanese\/."},{"key":"48","unstructured":"\u585a\u8d8a\u99ff\uff0c\u7b39\u91ce\u907c\u5e73 (2025). Ruri: \u65e5\u672c\u8a9e\u306b\u7279\u5316\u3057\u305f\u6c4e\u7528\u30c6\u30ad\u30b9\u30c8\u57cb\u3081\u8fbc\u307f\u30e2\u30c7\u30eb.  \u8a00\u8a9e\u51e6\u7406\u5b66\u4f1a \u7b2c31\u56de\u5e74\u6b21\u5927\u4f1a, pp. 1622\u20131627. [H. Tsukagoshi and R. Sasano (2025). Ruri: Nihongo ni Tokka shita Hanyo Tekisuto Umekomi Moderu. Proceedings of the 31st Annual Meeting of the Association for Natural Language Processing, pp. 1622\u20131627.]."},{"key":"49","unstructured":"Tsukagoshi, H., Li, S., Fukuchi, A., and Shibata, T. (2025). \u201cModernBERT-Ja.\u201d  https:\/\/huggingface.co\/collections\/sbintuitions\/modernbert-ja."},{"key":"50","unstructured":"Tsukagoshi, H., and Sasano, R. (2024). \u201cRuri: Japanese General Text Embeddings.\u201d  <i>arXiv preprint arXiv:2409.07737<\/i>."},{"key":"51","doi-asserted-by":"crossref","unstructured":"Tsukagoshi, H., and Sasano, R. (2025). \u201cRedundancy, Isotropy, and Intrinsic Dimensionality of Prompt-based Text Embeddings.\u201d  In <i>Findings of the Association for Computational Linguistics: ACL 2025<\/i>, pp. 25915\u201325930.","DOI":"10.18653\/v1\/2025.findings-acl.1330"},{"key":"52","unstructured":"van den Oord, A., Li, Y., and Vinyals, O. (2019). \u201cRepresentation Learning with Contrastive Predictive Coding.\u201d  <i>arXiv preprint arXiv:1807.03748<\/i>."},{"key":"53","unstructured":"Wang, L., Yang, N., Huang, X., Jiao, B., Yang, L., Jiang, D., Majumder, R., and Wei, F. (2022). \u201cText Embeddings by Weakly-Supervised Contrastive Pre-training.\u201d  <i>arXiv preprint arXiv:2212.03533<\/i>."},{"key":"54","doi-asserted-by":"crossref","unstructured":"Wang, L., Yang, N., Huang, X., Jiao, B., Yang, L., Jiang, D., Majumder, R., and Wei, F. (2023). \u201cSimLM: Pre-training with Representation Bottleneck for Dense Passage Retrieval.\u201d  In <i>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (ACL)<\/i>, pp. 2244\u20132258.","DOI":"10.18653\/v1\/2023.acl-long.125"},{"key":"55","doi-asserted-by":"crossref","unstructured":"Wang, L., Yang, N., Huang, X., Yang, L., Majumder, R., and Wei, F. (2024a). \u201cImproving Text Embeddings with Large Language Models.\u201d  In <i>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (ACL)<\/i>, pp. 11897\u201311916.","DOI":"10.18653\/v1\/2024.acl-long.642"},{"key":"56","unstructured":"Wang, L., Yang, N., Huang, X., Yang, L., Majumder, R., and Wei, F. (2024b). \u201cMultilingual E5 Text Embeddings: A Technical Report.\u201d  <i>arXiv preprint arXiv:2402.05672<\/i>."},{"key":"57","unstructured":"Wang, T., and Isola, P. (2020). \u201cUnderstanding Contrastive Representation Learning through Alignment and Uniformity on the Hypersphere.\u201d  In <i>Proceedings of the 37th International Conference on Machine Learning (ICML)<\/i>, pp. 9929\u20139939."},{"key":"58","doi-asserted-by":"crossref","unstructured":"Warner, B., Chaffin, A., Clavi\u00e9, B., Weller, O., Hallstr\u00f6m, O., Taghadouini, S., Gallagher, A., Biswas, R., Ladhak, F., Aarsen, T., Adams, G. T., Howard, J., and Poli, I. (2025). \u201cSmarter, Better, Faster, Longer: A Modern Bidirectional Encoder for Fast, Memory Efficient, and Long Context Finetuning and Inference.\u201d  In <i>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (ACL 2025)<\/i>, pp. 2526\u20132547.","DOI":"10.18653\/v1\/2025.acl-long.127"},{"key":"59","doi-asserted-by":"crossref","unstructured":"Williams, A., Nangia, N., and Bowman, S. (2018). \u201cA Broad-Coverage Challenge Corpus for Sentence Understanding through Inference.\u201d  In <i>Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL)<\/i>, pp. 1112\u20131122.","DOI":"10.18653\/v1\/N18-1101"},{"key":"60","unstructured":"Wortsman, M., Ilharco, G., Gadre, S. Y., Roelofs, R., Gontijo-Lopes, R., Morcos, A. S., Namkoong, H., Farhadi, A., Carmon, Y., Kornblith, S., and Schmidt, L. (2022). \u201cModel Soups: Averaging Weights of Multiple Fine-tuned Models Improves Accuracy without Increasing Inference Time.\u201d  In <i>Proceedings of the 39th International Conference on Machine Learning (ICML)<\/i>, pp. 23965\u201323998."},{"key":"61","doi-asserted-by":"crossref","unstructured":"Xiao, C., Long, Y., and Al Moubayed, N. (2023). \u201cOn Isotropy, Contextualization and Learning Dynamics of Contrastive-based Sentence Representation Learning.\u201d  In <i>Findings of the Association for Computational Linguistics: ACL 2023<\/i>, pp. 12266\u201312283.","DOI":"10.18653\/v1\/2023.findings-acl.778"},{"key":"62","doi-asserted-by":"crossref","unstructured":"Xiao, S., Liu, Z., Zhang, P., Muennighoff, N., Lian, D., and Nie, J.-Y. (2024). \u201cC-Pack: Packed Resources For General Chinese Embeddings.\u201d  In <i>Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)<\/i>, pp. 641\u2013649.","DOI":"10.1145\/3626772.3657878"},{"key":"63","doi-asserted-by":"crossref","unstructured":"Yanaka, H., and Mineshima, K. (2021). \u201cAssessing the Generalization Capacity of Pre-trained Language Models through Japanese Adversarial Natural Language Inference.\u201d  In <i>Proceedings of the 4th BlackboxNLP Workshop on Analyzing and Interpreting Neural Networks for NLP (BlackboxNLP)<\/i>, pp. 337\u2013349.","DOI":"10.18653\/v1\/2021.blackboxnlp-1.26"},{"key":"64","doi-asserted-by":"crossref","unstructured":"Yanaka, H., and Mineshima, K. (2022). \u201cCompositional Evaluation on Japanese Textual Entailment and Similarity.\u201d  <i>Transactions of the Association for Computational Linguistics (TACL)<\/i>, <b>10<\/b>, pp. 1266\u20131284.","DOI":"10.1162\/tacl_a_00518"},{"key":"65","doi-asserted-by":"crossref","unstructured":"Yang, Y., Zhang, Y., Tar, C., and Baldridge, J. (2019). \u201cPAWS-X: A Cross-lingual Adversarial Dataset for Paraphrase Identification.\u201d  In <i>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)<\/i>, pp. 3687\u20133692.","DOI":"10.18653\/v1\/D19-1382"},{"key":"66","unstructured":"\u5409\u8d8a\u5353\u898b\uff0c\u6cb3\u539f\u5927\u8f14\uff0c\u9ed2\u6a4b\u798e\u592b (2020). \u6a5f\u68b0\u7ffb\u8a33\u3092\u7528\u3044\u305f\u81ea\u7136\u8a00\u8a9e\u63a8\u8ad6\u30c7\u30fc\u30bf\u30bb\u30c3\u30c8\u306e\u591a\u8a00\u8a9e\u5316.  \u60c5\u5831\u51e6\u7406\u5b66\u4f1a \u7b2c244\u56de\u81ea\u7136\u8a00\u8a9e\u51e6\u7406\u7814\u7a76\u4f1a, 2020-NL-244 (6), pp. 1\u20138. [T. Yoshikoshi et al. (2020). Multilingualization of a Natural Language Inference Dataset Using Machine Translation. IPSJ SIG Technical Reports, 2020-NL-244, pp. 1\u20138.]."},{"key":"67","doi-asserted-by":"crossref","unstructured":"Zhang, J., Lan, Z., and He, J. (2023). \u201cContrastive Learning of Sentence Embeddings from Scratch.\u201d  In <i>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP 2023)<\/i>, pp. 3916\u20133932.","DOI":"10.18653\/v1\/2023.emnlp-main.238"},{"key":"68","doi-asserted-by":"crossref","unstructured":"Zhang, X., Ma, X., Shi, P., and Lin, J. (2021). \u201cMr. TyDi: A Multi-lingual Benchmark for Dense Retrieval.\u201d  In <i>Proceedings of the 1st Workshop on Multilingual Representation Learning (MRL)<\/i>, pp. 127\u2013137.","DOI":"10.18653\/v1\/2021.mrl-1.12"},{"key":"69","unstructured":"Zhang, X., Thakur, N., Ogundepo, O., Kamalloo, E., Alfonso-Hermelo, D., Li, X., Liu, Q., Rezagholizadeh, M., and Lin, J. (2022). \u201cMaking a MIRACL: Multilingual Information Retrieval Across a Continuum of Languages.\u201d  <i>arXiv preprint arXiv:2210.09984<\/i>."}],"container-title":["Journal of Natural Language Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_591\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:44:26Z","timestamp":1781930666000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_591\/_article\/-char\/ja\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":69,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.5715\/jnlp.33.591","relation":{},"ISSN":["1340-7619","2185-8314"],"issn-type":[{"value":"1340-7619","type":"print"},{"value":"2185-8314","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}