{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,1,18]],"date-time":"2025-01-18T05:06:08Z","timestamp":1737176768825,"version":"3.33.0"},"reference-count":27,"publisher":"Oxford University Press (OUP)","issue":"6","license":[{"start":{"date-parts":[[2024,4,5]],"date-time":"2024-04-05T00:00:00Z","timestamp":1712275200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/academic.oup.com\/pages\/standard-publication-reuse-rights"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,11,25]]},"abstract":"<jats:title>Abstract<\/jats:title>\n               <jats:p>We present PhrasIS, a benchmark dataset composed of natural occurring Phrase pairs with Inference and Similarity annotations for the evaluation of semantic representations. The described dataset fills the gap between word and sentence-level datasets, allowing to evaluate compositional models at a finer granularity than sentences. Contrary to other datasets, the phrase pairs are extracted from naturally occurring text in image captions and news headlines. All the text fragments have been annotated by experts following a rigorous process also described in the manuscript achieving high inter annotator agreement. In this work we analyse the dataset, showing the relation between inference labels and similarity scores. With 10K phrase pairs split in development and test, the dataset is an excellent benchmark for testing meaning representation systems.<\/jats:p>","DOI":"10.1093\/jigpal\/jzae037","type":"journal-article","created":{"date-parts":[[2024,4,6]],"date-time":"2024-04-06T06:04:46Z","timestamp":1712383486000},"page":"1088-1101","source":"Crossref","is-referenced-by-count":0,"title":["PhrasIS: Phrase Inference and Similarity benchmark"],"prefix":"10.1093","volume":"32","author":[{"given":"I","family":"Lopez-Gazpio","sequence":"first","affiliation":[{"name":"TIEC Department , University of Mundaiz kalea 50, Donostia 20012 , , inigo.lopez@ehu.eus","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J","family":"Gaviria","sequence":"additional","affiliation":[{"name":"TIEC Department , University of Mundaiz kalea 50, Donostia 20012 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"P","family":"Garc\u00eda","sequence":"additional","affiliation":[{"name":"TIEC Department , University of Mundaiz kalea 50, Donostia 20012 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"H","family":"Sanjurjo-Gonz\u00e1lez","sequence":"additional","affiliation":[{"name":"TIEC Department , University of Mundaiz kalea 50, Donostia 20012 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"B","family":"Sanz","sequence":"additional","affiliation":[{"name":"HiTZ Basque Center for Language Technologies\u2013Ixa NLP Group , University of the Basque Country (UPV\/EHU). M. Lardizabal 1, Donostia 20018 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"A","family":"Zarranz","sequence":"additional","affiliation":[{"name":"HiTZ Basque Center for Language Technologies\u2013Ixa NLP Group , University of the Basque Country (UPV\/EHU). M. Lardizabal 1, Donostia 20018 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M","family":"Maritxalar","sequence":"additional","affiliation":[{"name":"HiTZ Basque Center for Language Technologies\u2013Ixa NLP Group , University of the Basque Country (UPV\/EHU). M. Lardizabal 1, Donostia 20018 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"E","family":"Agirre","sequence":"additional","affiliation":[{"name":"HiTZ Basque Center for Language Technologies\u2013Ixa NLP Group , University of the Basque Country (UPV\/EHU). M. Lardizabal 1, Donostia 20018 ,","place":["Spain"]}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"286","published-online":{"date-parts":[[2024,4,5]]},"reference":[{"key":"2025011705415457000_ref1","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/S15-2045","article-title":"SemEval-2015 task 2: semantic textual similarity, English, Spanish and pilot on interpretability","volume-title":"Proceedings of the 9th International Workshop on Semantic Evaluation","author":"Agirre","year":"2015"},{"key":"2025011705415457000_ref2","first-page":"512","article-title":"Semeval-2016 task 2: interpretable semantic textual similarity","author":"Agirre","year":"2016","journal-title":"Proceedings of SemEval"},{"key":"2025011705415457000_ref3","first-page":"534","article-title":"Naturalli: natural logic inference for common sense reasoning","volume-title":"EMNLP","author":"Angeli","year":"2014"},{"key":"2025011705415457000_ref4","doi-asserted-by":"crossref","first-page":"95","DOI":"10.1007\/s10579-015-9332-5","article-title":"SICK through the semeval glasses","volume":"50","author":"Bentivogli","year":"2016","journal-title":"Language Resources and Evaluation"},{"key":"2025011705415457000_ref5","article-title":"Europe media monitor\u2014system description","volume-title":"EUR Report 22173-En","author":"Best","year":"2005"},{"key":"2025011705415457000_ref6","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/D15-1075","article-title":"A large annotated corpus for learning natural language inference","volume-title":"Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing (EMNLP)","author":"Bowman","year":"2015"},{"key":"2025011705415457000_ref7","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1017\/S1351324909990234","article-title":"Recognizing textual entailment: rational, evaluation and approaches","volume":"16","author":"Dagan","year":"2010","journal-title":"Natural Language Engineering"},{"key":"2025011705415457000_ref8","doi-asserted-by":"crossref","first-page":"350","DOI":"10.3115\/1220355.1220406","article-title":"Unsupervised construction of large paraphrase corpora: exploiting massively parallel news sources","volume-title":"COLING \u201904: Proceedings of the 20th International Conference on Computational Linguistics","author":"Dolan","year":"2004"},{"key":"2025011705415457000_ref9","first-page":"758","article-title":"Ppdb: the paraphrase database","volume-title":"Proceedings of the 2013 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"Ganitkevitch","year":"2013"},{"key":"2025011705415457000_ref10","doi-asserted-by":"crossref","first-page":"665","DOI":"10.1162\/COLI_a_00237","article-title":"Simlex-999: Evaluating semantic models with (genuine) similarity estimation","volume":"41","author":"Hill","year":"2015","journal-title":"Computational Linguistics"},{"key":"2025011705415457000_ref11","doi-asserted-by":"crossref","first-page":"17","DOI":"10.3115\/v1\/S14-2003","article-title":"Semeval-2014 task 3: cross-level semantic similarity","volume-title":"Proceedings of the 8th International Workshop on Semantic Evaluation (SemEval 2014)","author":"Jurgens","year":"2014"},{"key":"2025011705415457000_ref12","first-page":"39","article-title":"Semeval-2013 task 5: evaluating phrasal semantics","volume-title":"Joint Conference on Lexical and Computational Semantics (*SEM)","author":"Korkontzelos","year":"2013"},{"key":"2025011705415457000_ref13","doi-asserted-by":"crossref","first-page":"645","DOI":"10.1016\/j.engappai.2019.07.010","article-title":"A reproducible survey on word embeddings and ontology-based methods for word similarity: linear combinations outperform the state of the art","volume":"85","author":"Lastra-D\u00edaz","year":"2019","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"2025011705415457000_ref14","doi-asserted-by":"crossref","first-page":"24","DOI":"10.3115\/1621474.1621479","article-title":"Semeval-2007 task 06: word-sense disambiguation of prepositions","volume-title":"Proceedings of the Fourth International Workshop on Semantic Evaluations (SemEval-2007)","author":"Litkowski","year":"2007"},{"volume-title":"Natural Language Inference","year":"2009","author":"MacCartney","key":"2025011705415457000_ref15"},{"key":"2025011705415457000_ref16","doi-asserted-by":"crossref","first-page":"193","DOI":"10.3115\/1654536.1654575","article-title":"Natural logic for textual inference","volume-title":"Proceedings of the ACL-PASCAL Workshop on Textual Entailment and Paraphrasing","author":"MacCartney","year":"2007"},{"key":"2025011705415457000_ref17","doi-asserted-by":"crossref","first-page":"1388","DOI":"10.1111\/j.1551-6709.2010.01106.x","article-title":"Composition in distributional models of semantics","volume":"34","author":"Mitchell","year":"2010","journal-title":"Cognitive Science"},{"key":"2025011705415457000_ref18","first-page":"1512","article-title":"Adding semantics to data-driven paraphrasing","volume-title":"Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics","author":"Pavlick","year":"2015"},{"key":"2025011705415457000_ref19","first-page":"425","article-title":"Ppdb 2.0: better paraphrase ranking, fine-grained entailment relations, word embeddings, and style classification","volume-title":"Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing","author":"Pavlick","year":"2015"},{"key":"2025011705415457000_ref20","doi-asserted-by":"crossref","first-page":"38","DOI":"10.3115\/1614025.1614037","article-title":"Wordnet::Similarity: measuring the relatedness of concepts","volume-title":"Demonstration Papers at HLT-NAACL 2004","author":"Pedersen","year":"2004"},{"key":"2025011705415457000_ref21","first-page":"2825","article-title":"Scikit-learn: machine learning in python","volume":"12","author":"Pedregosa","year":"2011","journal-title":"The Journal of Machine Learning research"},{"key":"2025011705415457000_ref22","first-page":"139","article-title":"Collecting image annotations using Amazon\u2019s mechanical Turk","volume-title":"Proceedings of the NAACL HLT 2010 Workshop on Creating Speech and Language Data with Amazon\u2019s Mechanical Turk","author":"Rashtchian","year":"2010"},{"key":"2025011705415457000_ref23","doi-asserted-by":"crossref","first-page":"108","DOI":"10.18653\/v1\/S16-2013","article-title":"Adding context to semantic data-driven paraphrasing","volume-title":"Proceedings of the Fifth Joint Conference on Lexical and Computational Semantics","author":"Shwartz","year":"2016"},{"key":"2025011705415457000_ref24","first-page":"1556","article-title":"Improved semantic representations from tree-structured long short-term memory networks","volume-title":"Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing","author":"Tai","year":"2015"},{"key":"2025011705415457000_ref25","first-page":"127","article-title":"Introduction to the conll-2000 shared task: Chunking","volume-title":"Proceedings of the 2nd Workshop on Learning Language in Logic and the 4th Conference on Computational Natural Language Learning-Volume 7","author":"Tjong Kim Sang","year":"2000"},{"key":"2025011705415457000_ref26","doi-asserted-by":"crossref","first-page":"345","DOI":"10.1162\/tacl_a_00143","article-title":"From paraphrase database to compositional paraphrase model and back","volume":"3","author":"Wieting","year":"2015","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"2025011705415457000_ref27","first-page":"678","article-title":"Online learning of relaxed ccg grammars for parsing to logical form","volume-title":"Proceedings of EMNLP-CoNLL","author":"Zettlemoyer","year":"2007"}],"container-title":["Logic Journal of the IGPL"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/academic.oup.com\/jigpal\/article-pdf\/32\/6\/1088\/60929228\/jzae037.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/academic.oup.com\/jigpal\/article-pdf\/32\/6\/1088\/60929228\/jzae037.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T05:42:21Z","timestamp":1737092541000},"score":1,"resource":{"primary":{"URL":"https:\/\/academic.oup.com\/jigpal\/article\/32\/6\/1088\/7639121"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,5]]},"references-count":27,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2024,11,25]]}},"URL":"https:\/\/doi.org\/10.1093\/jigpal\/jzae037","relation":{},"ISSN":["1367-0751","1368-9894"],"issn-type":[{"type":"print","value":"1367-0751"},{"type":"electronic","value":"1368-9894"}],"subject":[],"published-other":{"date-parts":[[2024,12]]},"published":{"date-parts":[[2024,4,5]]}}}