{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T19:13:00Z","timestamp":1776107580619,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":137,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,4,19]],"date-time":"2023-04-19T00:00:00Z","timestamp":1681862400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,4,19]]},"DOI":"10.1145\/3544548.3580790","type":"proceedings-article","created":{"date-parts":[[2023,4,20]],"date-time":"2023-04-20T04:28:44Z","timestamp":1681964924000},"page":"1-20","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["Angler: Helping Machine Translation Practitioners Prioritize Model Improvements"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0386-4555","authenticated-orcid":false,"given":"Samantha","family":"Robertson","sequence":"first","affiliation":[{"name":"Electrical Engineering and Computer Sciences, University of California, Berkeley, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4360-1423","authenticated-orcid":false,"given":"Zijie J.","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computing, Georgia Tech, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3110-1053","authenticated-orcid":false,"given":"Dominik","family":"Moritz","sequence":"additional","affiliation":[{"name":"Apple, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1771-0565","authenticated-orcid":false,"given":"Mary Beth","family":"Kery","sequence":"additional","affiliation":[{"name":"Apple Inc., United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4164-844X","authenticated-orcid":false,"given":"Fred","family":"Hohman","sequence":"additional","affiliation":[{"name":"Apple, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,4,19]]},"reference":[{"key":"e_1_3_3_3_1_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-8659.2009.01443.x"},{"key":"e_1_3_3_3_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-SEIP.2019.00042"},{"key":"e_1_3_3_3_3_1","doi-asserted-by":"publisher","DOI":"10.1257\/pandp.20181003"},{"key":"e_1_3_3_3_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_5_1","first-page":"1","article-title":"Distribution-Matching Embedding for Visual Domain Adaptation","volume":"17","author":"Baktashmotlagh Mahsa","year":"2016","unstructured":"Mahsa Baktashmotlagh, Mehrtash Har, i, and Mathieu Salzmann. 2016. Distribution-Matching Embedding for Visual Domain Adaptation. Journal of Machine Learning Research 17, 108 (2016), 1\u201330. http:\/\/jmlr.org\/papers\/v17\/15-207.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_3_3_6_1","volume-title":"Proceedings of the Acl Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization. Association for Computational Linguistics","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In Proceedings of the Acl Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization. Association for Computational Linguistics, Ann Arbor, Michigan, 65\u201372."},{"key":"e_1_3_3_3_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3461702.3462610"},{"key":"e_1_3_3_3_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/mcg.2021.3107875"},{"key":"e_1_3_3_3_9_1","doi-asserted-by":"publisher","unstructured":"Jon\u00a0Louis Bentley. 1975. Multidimensional Binary Search Trees Used for Associative Searching. Commun. ACM 18(1975). https:\/\/doi.org\/10.1145\/361002.361007","DOI":"10.1145\/361002.361007"},{"key":"e_1_3_3_3_10_1","volume-title":"Workshop on Human Evaluation of NLP Systems.","author":"Bhatt Shaily","year":"2021","unstructured":"Shaily Bhatt, Rahul Jain, Sandipan Dandapat, and Sunayana Sitaram. 2021. A Case Study of Efficacy and Challenges in Practical Human-in-Loop Evaluation of NLP Systems Using Checklist. In Workshop on Human Evaluation of NLP Systems."},{"key":"e_1_3_3_3_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_12_1","unstructured":"Tolga Bolukbasi Kai-Wei Chang James\u00a0Y Zou Venkatesh Saligrama and Adam\u00a0T Kalai. 2016. Man Is to Computer Programmer as Woman Is to Homemaker? Debiasing Word Embeddings. In NeurIPS Vol.\u00a029."},{"key":"e_1_3_3_3_13_1","doi-asserted-by":"publisher","unstructured":"M. Bostock V. Ogievetsky and J. Heer. 2011. D3 Data-Driven Documents. IEEE TVCG 17 (Dec. 2011). https:\/\/doi.org\/10.1109\/tvcg.2011.185","DOI":"10.1109\/tvcg.2011.185"},{"key":"e_1_3_3_3_14_1","volume-title":"Proceedings of the 1st Conference on Fairness, Accountability and Transparency(Proceedings of Machine Learning Research, Vol.\u00a081)","author":"Buolamwini Joy","year":"2018","unstructured":"Joy Buolamwini and Timnit Gebru. 2018. Gender Shades: Intersectional Accuracy Disparities in Commercial Gender Classification. In Proceedings of the 1st Conference on Fairness, Accountability and Transparency(Proceedings of Machine Learning Research, Vol.\u00a081)."},{"key":"e_1_3_3_3_15_1","doi-asserted-by":"publisher","DOI":"10.1515\/pralin-2017-0017"},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3479569"},{"key":"e_1_3_3_3_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/vast47406.2019.8986948"},{"key":"e_1_3_3_3_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300234"},{"key":"e_1_3_3_3_21_1","volume-title":"Mechanical Turk. In Proceedings of the 2009 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics","author":"Callison-Burch Chris","year":"2009","unstructured":"Chris Callison-Burch. 2009. Fast, Cheap, and Creative: Evaluating Translation Quality Using Amazon\u2019s Mechanical Turk. In Proceedings of the 2009 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics, Singapore, 286\u2013295. https:\/\/aclanthology.org\/D09-1030"},{"key":"e_1_3_3_3_22_1","doi-asserted-by":"publisher","DOI":"10.3115\/1626355.1626373"},{"key":"e_1_3_3_3_23_1","volume-title":"Re-Evaluating the Role of Bleu in Machine Translation Research. In 11th Conference of the European Chapter of the Association for Computational Linguistics.","author":"Callison-Burch Chris","year":"2006","unstructured":"Chris Callison-Burch, Miles Osborne, and Philipp Koehn. 2006. Re-Evaluating the Role of Bleu in Machine Translation Research. In 11th Conference of the European Chapter of the Association for Computational Linguistics."},{"key":"e_1_3_3_3_24_1","doi-asserted-by":"publisher","unstructured":"Leshem Choshen and Omri Abend. 2019. Automatically Extracting Challenge Sets for Non-Local Phenomena in Neural Machine Translation. In CoNLL. https:\/\/doi.org\/10.18653\/v1\/k19-1028","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2019.2916074"},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/1456650.1456652"},{"key":"e_1_3_3_3_27_1","unstructured":"Andy Coenen and Adam Pearce. 2019. Understanding UMAP. https:\/\/pair-code.github.io\/understanding-umap\/"},{"key":"e_1_3_3_3_28_1","volume-title":"Dangers of Machine Translation: The Need for Professionally Translated Anticipatory Guidance Resources for Limited English Proficiency Caregivers. Clinical Pediatrics (Phila) 58 (Feb","author":"Das Prithwijit","year":"2019","unstructured":"Prithwijit Das, Anna Kuznetsova, Meng\u2019ou Zhu, and Ruth Milanaik. 2019. Dangers of Machine Translation: The Need for Professionally Translated Anticipatory Guidance Resources for Limited English Proficiency Caregivers. Clinical Pediatrics (Phila) 58 (Feb. 2019)."},{"key":"e_1_3_3_3_29_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops.","author":"de Vries Terrance","unstructured":"Terrance de Vries, Ishan Misra, Changhan Wang, and Laurens van der Maaten. 2019. Does Object Recognition Work for Everyone?. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops."},{"key":"e_1_3_3_3_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533240"},{"key":"e_1_3_3_3_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517441"},{"key":"e_1_3_3_3_32_1","volume-title":"Fairlearn: A Toolkit for Assessing and Improving Fairness in AI. (May","author":"Dud\u00edk Miro","year":"2020","unstructured":"Miro Dud\u00edk, Sarah Bird, Hanna Wallach, and Kathleen Walker. 2020. Fairlearn: A Toolkit for Assessing and Improving Fairness in AI. (May 2020)."},{"key":"e_1_3_3_3_33_1","volume-title":"Helena Moniz, and Andr\u00e9 Martins.","author":"Farinha C.","year":"2022","unstructured":"Ana\u00a0C. Farinha, M.\u00a0Amin Farajian, Patrick Fernandes Jos\u00e9 Souza Jo\u00e3o Alves\u00a0Ant\u00f3nio Lopes, Helena Moniz, and Andr\u00e9 Martins. 2022. WMT22 Chat Task. https:\/\/wmt-chat-task.github.io\/"},{"key":"e_1_3_3_3_34_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00437"},{"key":"e_1_3_3_3_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.117"},{"key":"e_1_3_3_3_36_1","doi-asserted-by":"publisher","unstructured":"Satvik Garg Pradyumn Pundir Geetanjali Rathee P.K. Gupta Somya Garg and Saransh Ahlawat. 2021. On Continuous Integration \/ Continuous Delivery for Automated Deployment of Machine Learning Models Using MLOps. In 2021 IEEE Fourth International Conference on Artificial Intelligence and Knowledge Engineering (AIKE). https:\/\/doi.org\/10.1109\/aike52691.2021.00010","DOI":"10.1109\/aike52691.2021.00010"},{"key":"e_1_3_3_3_37_1","doi-asserted-by":"publisher","unstructured":"Mor Geva Yoav Goldberg and Jonathan Berant. 2019. Are We Modeling the Task or the Annotator? An Investigation of Annotator Bias in Natural Language Understanding Datasets. In EMNLP-IJCNLP. https:\/\/doi.org\/10.18653\/v1\/d19-1107","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_38_1","unstructured":"Maarten Grootendorst. 2022. BERTopic: Neural Topic Modeling with a Class-Based TF-IDF Procedure. arXiv preprint arXiv:2203.05794(2022)."},{"key":"e_1_3_3_3_39_1","volume-title":"Machine Translation Evaluation Resources and Methods: A Survey. arxiv:1605.04515 (Sept","author":"Han Lifeng","year":"2018","unstructured":"Lifeng Han. 2018. Machine Translation Evaluation Resources and Methods: A Survey. arxiv:1605.04515 (Sept. 2018)."},{"key":"e_1_3_3_3_40_1","volume-title":"Data Documentation Perceptions, Needs, Challenges, and Desiderata. arxiv:2206.02923 (Aug.","author":"Heger K.","year":"2022","unstructured":"Amy\u00a0K. Heger, Liz\u00a0B. Marquis, Mihaela Vorvoreanu, Hanna Wallach, and Jennifer\u00a0Wortman Vaughan. 2022. Understanding Machine Learning Practitioners\u2019 Data Documentation Perceptions, Needs, Challenges, and Desiderata. arxiv:2206.02923 (Aug. 2022)."},{"key":"e_1_3_3_3_41_1","doi-asserted-by":"publisher","DOI":"10.1515\/pralin-2017-0001"},{"key":"e_1_3_3_3_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2018.2843369"},{"key":"e_1_3_3_3_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300830"},{"key":"e_1_3_3_3_44_1","volume-title":"Characterising Bias in Compressed Models. arxiv:2010.03058 (Dec","author":"Hooker Sara","year":"2020","unstructured":"Sara Hooker, Nyalleng Moorosi, Gregory Clark, Samy Bengio, and Emily Denton. 2020. Characterising Bias in Compressed Models. arxiv:2010.03058 (Dec. 2020)."},{"key":"e_1_3_3_3_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_46_1","unstructured":"Aspen Hopkins Fred Hohman Luca Zappella Xavier\u00a0Suau Cuadros and Dominik Moritz. 2023. Designing data: Proactive data collection and iteration for machine learning. arXiv preprint arXiv:2301.10319(2023)."},{"key":"e_1_3_3_3_47_1","volume-title":"Designing Machine Learning Systems: An Iterative Process for Production-Ready Applications","author":"Huyen Chip","unstructured":"Chip Huyen. 2022. Designing Machine Learning Systems: An Iterative Process for Production-Ready Applications (first edition ed.)."},{"key":"e_1_3_3_3_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/1837885.1837906"},{"key":"e_1_3_3_3_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_50_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00065"},{"key":"e_1_3_3_3_51_1","volume-title":"Fair SA: Sensitivity Analysis for Fairness in Face Recognition. In Algorithmic Fairness through the Lens of Causality and Robustness Workshop. PMLR.","author":"Joshi R","year":"2022","unstructured":"Aparna\u00a0R Joshi, Xavier\u00a0Suau Cuadros, Nivedha Sivakumar, Luca Zappella, and Nicholas Apostoloff. 2022. Fair SA: Sensitivity Analysis for Fairness in Face Recognition. In Algorithmic Fairness through the Lens of Causality and Robustness Workshop. PMLR."},{"key":"e_1_3_3_3_52_1","doi-asserted-by":"publisher","DOI":"10.1001\/jamainternmed.2018.7653"},{"key":"e_1_3_3_3_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_54_1","unstructured":"Margaret King and Bente Maeggard. 1998. Issues in Natural Language Systems Evaluation.. In LREC."},{"key":"e_1_3_3_3_55_1","volume-title":"Comparative Evaluation of Online Machine Translation Systems with Legal Texts","author":"Kit Chunyu","year":"2008","unstructured":"Chunyu Kit and Tak\u00a0Ming Wong. 2008. Comparative Evaluation of Online Machine Translation Systems with Legal Texts. Law Library Journal 100(2008)."},{"key":"e_1_3_3_3_56_1","doi-asserted-by":"publisher","DOI":"10.1515\/pralin-2015-0014"},{"key":"e_1_3_3_3_57_1","volume-title":"Proceedings of Machine Translation Summit XI: Invited Papers.","author":"Koehn Philipp","year":"2007","unstructured":"Philipp Koehn. 09 10-14 2007. EuroMatrix \u2013 Machine Translation for All European Languages. In Proceedings of Machine Translation Summit XI: Invited Papers."},{"key":"e_1_3_3_3_58_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1915768117"},{"key":"e_1_3_3_3_59_1","volume-title":"WILDS: A Benchmark of in-the-Wild Distribution Shifts. In ICML(Proceedings of Machine Learning Research, Vol.\u00a0139).","author":"Koh Pang\u00a0Wei","year":"2021","unstructured":"Pang\u00a0Wei Koh, Shiori Sagawa, Henrik Marklund, Sang\u00a0Michael Xie, Marvin Zhang, Akshay Balsubramani, Weihua Hu, Michihiro Yasunaga, Richard\u00a0Lanas Phillips, Irena Gao, Tony Lee, Etienne David, Ian Stavness, Wei Guo, Berton Earnshaw, Imran Haque, Sara\u00a0M Beery, Jure Leskovec, Anshul Kundaje, Emma Pierson, Sergey Levine, Chelsea Finn, and Percy Liang. 2021. WILDS: A Benchmark of in-the-Wild Distribution Shifts. In ICML(Proceedings of Machine Learning Research, Vol.\u00a0139)."},{"key":"e_1_3_3_3_60_1","volume-title":"UDIS: Unsupervised Discovery of Bias in Deep Visual Recognition Models. arxiv:2110.15499 (Oct.","author":"Krishnakumar Arvindkumar","year":"2021","unstructured":"Arvindkumar Krishnakumar, Viraj Prabhu, Sruthi Sudhakar, and Judy Hoffman. 2021. UDIS: Unsupervised Discovery of Bias in Deep Visual Recognition Models. arxiv:2110.15499 (Oct. 2021)."},{"key":"e_1_3_3_3_61_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-2021"},{"key":"e_1_3_3_3_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02081"},{"key":"e_1_3_3_3_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2022.3184186"},{"key":"e_1_3_3_3_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376261"},{"key":"e_1_3_3_3_65_1","doi-asserted-by":"publisher","DOI":"10.3115\/1220355.1220427"},{"key":"e_1_3_3_3_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2017.2744378"},{"key":"e_1_3_3_3_67_1","doi-asserted-by":"publisher","DOI":"10.3115\/1219044.1219075"},{"key":"e_1_3_3_3_68_1","volume-title":"Unit Testing for Concepts in Neural Networks. arxiv:2208.10244 (July","author":"Lovering Charles","year":"2022","unstructured":"Charles Lovering and Ellie Pavlick. 2022. Unit Testing for Concepts in Neural Networks. arxiv:2208.10244 (July 2022)."},{"key":"e_1_3_3_3_69_1","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems(NIPS\u201917)","author":"M.","unstructured":"Scott\u00a0M. Lundberg and Su-In Lee. 2017. A Unified Approach to Interpreting Model Predictions. In Proceedings of the 31st International Conference on Neural Information Processing Systems(NIPS\u201917)."},{"key":"e_1_3_3_3_70_1","doi-asserted-by":"publisher","unstructured":"Hossin M and Sulaiman M.N. 2015. A Review on Evaluation Metrics for Data Classification Evaluations. International Journal of Data Mining & Knowledge Management Process 5 (March 2015). https:\/\/doi.org\/10.5121\/ijdkp.2015.5201","DOI":"10.5121\/ijdkp.2015.5201"},{"key":"e_1_3_3_3_71_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/icsc.2011.36"},{"key":"e_1_3_3_3_73_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_74_1","volume-title":"UMAP: Uniform Manifold Approximation and Projection for Dimension Reduction. ArXiv e-prints (Feb.","author":"McInnes L.","year":"2018","unstructured":"L. McInnes, J. Healy, and J. Melville. 2018. UMAP: Uniform Manifold Approximation and Projection for Dimension Reduction. ArXiv e-prints (Feb. 2018)."},{"key":"e_1_3_3_3_75_1","volume-title":"Qualitative research in practice: Examples for discussion and analysis. Jossey-Bass","author":"Merriam B","unstructured":"Sharan\u00a0B Merriam and Associates. 2002. Introduction to qualitative research. In Qualitative research in practice: Examples for discussion and analysis. Jossey-Bass, Hoboken, NJ, USA, 1\u201317."},{"key":"e_1_3_3_3_76_1","volume-title":"Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC\u201914)","author":"Mor\u00e9 Joaquim","year":"2014","unstructured":"Joaquim Mor\u00e9 and Salvador Climent. 2014. Machine Translationness: Machine-likeness in Machine Translation Evaluation. In Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC\u201914)."},{"key":"e_1_3_3_3_77_1","volume-title":"Visual Auditor: Interactive Visualization for Detection and Summarization of Model Biases. In 2022 IEEE Visualization Conference (VIS).","author":"Munechika David","year":"2022","unstructured":"David Munechika, Zijie\u00a0J. Wang, Jack Reidy, Josh Rubin, Krishna Gade, Krishnaram Kenthapadi, and Duen\u00a0Horng Chau. 2022. Visual Auditor: Interactive Visualization for Detection and Summarization of Model Biases. In 2022 IEEE Visualization Conference (VIS)."},{"key":"e_1_3_3_3_78_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cag.2021.12.003"},{"key":"e_1_3_3_3_79_1","volume-title":"Stress Test Evaluation for Natural Language Inference. In The 27th International Conference on Computational Linguistics (COLING).","author":"Naik Aakanksha","year":"2018","unstructured":"Aakanksha Naik, Abhilasha Ravichander, Norman Sadeh, Carolyn Rose, and Graham Neubig. 2018. Stress Test Evaluation for Natural Language Inference. In The 27th International Conference on Computational Linguistics (COLING)."},{"key":"e_1_3_3_3_80_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_81_1","doi-asserted-by":"publisher","DOI":"10.1126\/science.aax2342"},{"key":"e_1_3_3_3_82_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2020.3030361"},{"key":"e_1_3_3_3_83_1","volume-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics.","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: A Method for Automatic Evaluation of Machine Translation. In Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_3_3_84_1","doi-asserted-by":"publisher","DOI":"10.1109\/pacificvis53943.2022.00029"},{"key":"e_1_3_3_3_85_1","doi-asserted-by":"publisher","DOI":"10.1145\/1357054.1357160"},{"key":"e_1_3_3_3_86_1","doi-asserted-by":"publisher","unstructured":"Karl Pearson. 1901. On Lines and Planes of Closest Fit to Systems of Points in Space. The London Edinburgh and Dublin Philosophical Magazine and Journal of Science 2(1901). https:\/\/doi.org\/10.1080\/14786440109462720","DOI":"10.1080\/14786440109462720"},{"key":"e_1_3_3_3_87_1","volume-title":"Scikit-Learn: Machine Learning in Python. the Journal of machine Learning research 12","author":"Pedregosa Fabian","year":"2011","unstructured":"Fabian Pedregosa, Ga\u00ebl Varoquaux, Alexandre Gramfort, Vincent Michel, Bertrand Thirion, Olivier Grisel, Mathieu Blondel, Peter Prettenhofer, Ron Weiss, Vincent Dubourg, 2011. Scikit-Learn: Machine Learning in Python. the Journal of machine Learning research 12 (2011)."},{"key":"e_1_3_3_3_88_1","volume-title":"Proceedings of Machine Learning and Systems, Vol.\u00a01.","author":"Polyzotis Neoklis","year":"2019","unstructured":"Neoklis Polyzotis, Martin Zinkevich, Sudip Roy, Eric Breck, and Steven Whang. 2019. Data Validation for Machine Learning. In Proceedings of Machine Learning and Systems, Vol.\u00a01."},{"key":"e_1_3_3_3_89_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_90_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_91_1","volume-title":"Proceedings of Machine Translation Summit XVII: Tutorial Abstracts.","author":"Popovi\u0107 Maja","year":"2019","unstructured":"Maja Popovi\u0107 and Sheila Castilho. 2019. Challenge Test Sets for MT Evaluation. In Proceedings of Machine Translation Summit XVII: Tutorial Abstracts."},{"key":"e_1_3_3_3_92_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_93_1","volume-title":"Androcentrism and Back-Translation Foibles. In AfricaNLP Workshop at EACL.","author":"Prabhu Vinay","year":"2021","unstructured":"Vinay Prabhu, Ryan Teehan, Eniko Srivastava, and Abdul Nimeri. 2021. Did They Direct the Violence or Admonish It? A Cautionary Tale on Contronomy, Androcentrism and Back-Translation Foibles. In AfricaNLP Workshop at EACL."},{"key":"e_1_3_3_3_94_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0139)","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models from Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0139). https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"e_1_3_3_3_95_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_96_1","doi-asserted-by":"publisher","DOI":"10.1145\/3375627.3375820"},{"key":"e_1_3_3_3_97_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_98_1","unstructured":"Sylvestre-Alvise Rebuffi Sven Gowal Dan\u00a0Andrei Calian Florian Stimberg Olivia Wiles and Timothy\u00a0A Mann. 2021. Data Augmentation Can Improve Robustness. In Advances in Neural Information Processing Systems Vol.\u00a034."},{"key":"e_1_3_3_3_99_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097)","author":"Recht Benjamin","year":"2019","unstructured":"Benjamin Recht, Rebecca Roelofs, Ludwig Schmidt, and Vaishaal Shankar. 2019. Do ImageNet Classifiers Generalize to ImageNet?. In Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097)."},{"key":"e_1_3_3_3_100_1","doi-asserted-by":"publisher","unstructured":"Nils Reimers and Iryna Gurevych. 2019. Sentence-BERT: Sentence Embeddings Using Siamese BERT-Networks. In EMNLP. https:\/\/doi.org\/10.18653\/v1\/d19-1410","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_101_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00322"},{"key":"e_1_3_3_3_102_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.230"},{"key":"e_1_3_3_3_103_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939778"},{"key":"e_1_3_3_3_104_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.442"},{"key":"e_1_3_3_3_105_1","doi-asserted-by":"publisher","DOI":"10.1515\/pralin-2017-0037"},{"key":"e_1_3_3_3_106_1","doi-asserted-by":"publisher","unstructured":"Paul R\u00f6ttger Bertie Vidgen Dong Nguyen Zeerak Waseem Helen Margetts and Janet Pierrehumbert. 2021. HateCheck: Functional Tests for Hate Speech Detection Models. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.4","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_107_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2002"},{"key":"e_1_3_3_3_108_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485766"},{"key":"e_1_3_3_3_109_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_110_1","doi-asserted-by":"publisher","DOI":"10.1145\/3299869.3320210"},{"key":"e_1_3_3_3_111_1","doi-asserted-by":"publisher","unstructured":"David\u00a0W. Scott. 2015. Multivariate Density Estimation: Theory Practice and Visualization. https:\/\/doi.org\/10.1002\/9781118575574","DOI":"10.1002\/9781118575574"},{"key":"e_1_3_3_3_112_1","unstructured":"D. Sculley. 2022. A Data-Centric View of Technical Debt in AI. https:\/\/datacentricai.org\/data-in-deployment\/"},{"key":"e_1_3_3_3_113_1","unstructured":"D. Sculley Gary Holt Daniel Golovin Eugene Davydov Todd Phillips Dietmar Ebner Vinay Chaudhary Michael Young Jean-Fran\u00e7ois Crespo and Dan Dennison. 2015. Hidden Technical Debt in Machine Learning Systems. In Advances in Neural Information Processing Systems Vol.\u00a028."},{"key":"e_1_3_3_3_114_1","doi-asserted-by":"publisher","DOI":"10.1145\/3479577"},{"key":"e_1_3_3_3_115_1","doi-asserted-by":"publisher","unstructured":"Bernard\u00a0W Silverman. 2018. Density Estimation for Statistics and Data Analysis. https:\/\/doi.org\/10.1201\/9781315140919","DOI":"10.1201\/9781315140919"},{"key":"e_1_3_3_3_116_1","volume-title":"Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC-2018)","author":"Soares Felipe","year":"2018","unstructured":"Felipe Soares, Viviane Moreira, and Karin Becker. 2018. A Large Parallel Corpus of Full-Text Scientific Articles. In Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC-2018)(Miyazaki, Japan). European Language Resource Association. http:\/\/aclweb.org\/anthology\/L18-1546"},{"key":"e_1_3_3_3_117_1","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2019.2934629"},{"key":"e_1_3_3_3_118_1","doi-asserted-by":"publisher","unstructured":"Gabriel Stanovsky Noah\u00a0A. Smith and Luke Zettlemoyer. 2019. Evaluating Gender Bias in Machine Translation. In ACL. https:\/\/doi.org\/10.18653\/v1\/p19-1164","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_119_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-5015"},{"key":"e_1_3_3_3_120_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30586-6_38"},{"key":"e_1_3_3_3_121_1","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2018.2865044"},{"key":"e_1_3_3_3_122_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11606-021-06666-z"},{"key":"e_1_3_3_3_123_1","volume-title":"Proceedings of the Fifth Conference on Machine Translation. Association for Computational Linguistics, Online, 1174\u20131182","author":"Tiedemann J\u00f6rg","year":"2020","unstructured":"J\u00f6rg Tiedemann. 2020. The Tatoeba Translation Challenge \u2013 Realistic Data Sets for Low Resource and Multilingual MT. In Proceedings of the Fifth Conference on Machine Translation. Association for Computational Linguistics, Online, 1174\u20131182. https:\/\/www.aclweb.org\/anthology\/2020.wmt-1.139"},{"key":"e_1_3_3_3_124_1","volume-title":"Proceedings of the 22nd Annual Conferenec of the European Association for Machine Translation (EAMT).","author":"Tiedemann J\u00f6rg","year":"2020","unstructured":"J\u00f6rg Tiedemann and Santhosh Thottingal. 2020. OPUS-MT \u2014 Building Open Translation Services for the World. In Proceedings of the 22nd Annual Conferenec of the European Association for Machine Translation (EAMT)."},{"key":"e_1_3_3_3_125_1","volume-title":"Proceedings of the Sixth Conference on Machine Translation.","author":"Troles Jonas-Dario","year":"2021","unstructured":"Jonas-Dario Troles and Ute Schmid. 2021. Extending Challenge Sets to Uncover Gender Bias in Machine Translation: Impact of Stereotypical Verbs and Adjectives. In Proceedings of the Sixth Conference on Machine Translation."},{"key":"e_1_3_3_3_126_1","volume-title":"The Visual Display of Quantitative Information","author":"Tufte R.","unstructured":"Edward\u00a0R. Tufte. 2013. The Visual Display of Quantitative Information (2nd ed., 8th print ed.).","edition":"2"},{"key":"e_1_3_3_3_127_1","volume-title":"Visualizing Data Using T-SNE. Journal of Machine Learning Research 9","author":"van der Maaten Laurens","year":"2008","unstructured":"Laurens van der Maaten and Geoffrey Hinton. 2008. Visualizing Data Using T-SNE. Journal of Machine Learning Research 9 (2008). http:\/\/jmlr.org\/papers\/v9\/vandermaaten08a.html"},{"key":"e_1_3_3_3_128_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_129_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.saa.2021.119547"},{"key":"e_1_3_3_3_130_1","volume-title":"Proceedings of the Fifth International Conference on Language Resources and Evaluation (LREC\u201906)","author":"Vilar David","year":"2006","unstructured":"David Vilar, Jia Xu, Luis\u00a0Fernando D\u2019Haro, and Hermann Ney. 2006. Error Analysis of Statistical Machine Translation Output. In Proceedings of the Fifth International Conference on Language Resources and Evaluation (LREC\u201906)."},{"key":"e_1_3_3_3_131_1","doi-asserted-by":"publisher","unstructured":"Changhan Wang Anirudh Jain Danlu Chen and Jiatao Gu. 2019. VizSeq: A Visual Analysis Toolkit for Text Generation Tasks. In Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP): System Demonstrations. https:\/\/doi.org\/10.18653\/v1\/d19-3043","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_132_1","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2019.2934619"},{"key":"e_1_3_3_3_133_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_134_1","doi-asserted-by":"publisher","unstructured":"Tongshuang Wu Marco\u00a0Tulio Ribeiro Jeffrey Heer and Daniel Weld. 2021. Polyjuice: Generating Counterfactuals for Explaining Evaluating and Improving Models. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.523","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_135_1","volume-title":"Scaling the Tower of Babel Fish: An Analysis of the Machine Translation of Legal Information","author":"Yates Sarah","year":"2006","unstructured":"Sarah Yates. 2006. Scaling the Tower of Babel Fish: An Analysis of the Machine Translation of Legal Information. Law Library Journal 98(2006)."},{"key":"e_1_3_3_3_136_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2022.3209465"},{"key":"e_1_3_3_3_137_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2003"}],"event":{"name":"CHI '23: CHI Conference on Human Factors in Computing Systems","location":"Hamburg Germany","acronym":"CHI '23","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3580790","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3544548.3580790","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:37:25Z","timestamp":1750178245000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3580790"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,19]]},"references-count":137,"alternative-id":["10.1145\/3544548.3580790","10.1145\/3544548"],"URL":"https:\/\/doi.org\/10.1145\/3544548.3580790","relation":{},"subject":[],"published":{"date-parts":[[2023,4,19]]},"assertion":[{"value":"2023-04-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}